CANP-WP-0002 T04: eval schema, render checks and output criteria
`evals/` was a reserved path holding unvalidated blobs: section 12 named the directory and gave an illustrative snippet, but nothing was specified, so no tool could act on an eval file. Every eval file must now declare a `schema`, and CPF defines exactly one — `canned-prompts/eval-rubric/v0.1`. Unrecognized schemas stay legal and are skipped rather than rejected, so the format gains something actionable without becoming an evaluation language, which remains a non-goal. The schema splits along the same seam as T01 and T03. Render checks (`contains`, `not_contains`, `resolves_all`) assert properties of the rendered prompt text, need no model, and are therefore run by the reference CLI. Output criteria describe a good result and are declared but not run, because judging them requires a model. That division is now the format's consistent answer to "deterministic locally, or not". An eval references a fixture already declared in the manifest's `examples` rather than carrying its own copy, so an example that is also an eval fixture stays honest — both break together. An eval declares assessment and must not record outcomes. Results are run evidence and live outside the immutable package, per INTENT.md and section 17. Spec: 12 rewritten with 12.1, 18 (rules 17-18), 21 (`eval` verb). Reference CLI: read_eval, validate_eval, load_example_values, run_render_checks, cmd_eval; a failed render check exits non-zero. Tests 42 -> 51. examples/pqrst-estimate/evals/quality.yaml is a real eval with four render checks and four output criteria, and it passes. Fixes a latent bug reaching a fixture exposed: coerce_value assumed every value was a command-line string, so a YAML fixture carrying a real type (include_rationale: true) crashed on .lower(). Typed values are now validated but not re-parsed. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Bjefh8NUiEiahN4JLwoSKM Assistant: claude-code Assistant-Model: opus Assistant-Process: 388925@bnt-lap001 Assistant-Session: 3507023f-e0fd-4a1e-9d90-a0d4217d1502
This commit is contained in:
parent
cac6bc135f
commit
c5f1640454
8 changed files with 444 additions and 18 deletions
|
|
@ -504,3 +504,114 @@ inputs:
|
|||
)
|
||||
with pytest.raises(cp.CannedPromptError, match="must declare a version"):
|
||||
cp.validate_package(pkg)
|
||||
|
||||
|
||||
# --- evals (§ 12) ---
|
||||
|
||||
EVAL_BASE = """\
|
||||
format: canned-prompt/v0.1
|
||||
id: demo/evaluated
|
||||
name: Evaluated
|
||||
version: 1.0.0
|
||||
summary: Exercise eval files.
|
||||
template: prompt.md
|
||||
examples:
|
||||
- examples/basic.yaml
|
||||
evals:
|
||||
- evals/quality.yaml
|
||||
"""
|
||||
|
||||
|
||||
def write_evaluated(pkg: Path, eval_body: str, template: str = "Sum to 100%. {{ topic }}\n") -> Path:
|
||||
pkg.mkdir(exist_ok=True)
|
||||
(pkg / "prompt.yaml").write_text(
|
||||
EVAL_BASE + "inputs:\n - name: topic\n required: false\n default: cats\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(pkg / "prompt.md").write_text(template, encoding="utf-8")
|
||||
(pkg / "examples").mkdir(exist_ok=True)
|
||||
(pkg / "examples" / "basic.yaml").write_text(
|
||||
"name: basic\nvalues:\n topic: dogs\n", encoding="utf-8"
|
||||
)
|
||||
(pkg / "evals").mkdir(exist_ok=True)
|
||||
(pkg / "evals" / "quality.yaml").write_text(eval_body, encoding="utf-8")
|
||||
return pkg
|
||||
|
||||
|
||||
RUBRIC = """\
|
||||
schema: canned-prompts/eval-rubric/v0.1
|
||||
name: quality
|
||||
example: examples/basic.yaml
|
||||
render:
|
||||
- contains: "Sum to 100%"
|
||||
- not_contains: "{{"
|
||||
- resolves_all: true
|
||||
output:
|
||||
criteria:
|
||||
- Answers the question.
|
||||
"""
|
||||
|
||||
|
||||
def test_valid_eval_passes_validation(tmp_path: Path) -> None:
|
||||
pkg = write_evaluated(tmp_path / "p", RUBRIC)
|
||||
assert cp.validate_package(pkg)["id"] == "demo/evaluated"
|
||||
|
||||
|
||||
def test_eval_must_declare_a_schema(tmp_path: Path) -> None:
|
||||
pkg = write_evaluated(tmp_path / "p", "name: quality\nrender: []\n")
|
||||
with pytest.raises(cp.CannedPromptError, match="must declare a schema"):
|
||||
cp.validate_package(pkg)
|
||||
|
||||
|
||||
def test_unknown_eval_schema_is_ignored(tmp_path: Path) -> None:
|
||||
pkg = write_evaluated(tmp_path / "p", "schema: someone/else/v1\nwhatever: true\n")
|
||||
assert cp.validate_package(pkg)["id"] == "demo/evaluated"
|
||||
|
||||
|
||||
def test_eval_asserting_nothing_is_rejected(tmp_path: Path) -> None:
|
||||
pkg = write_evaluated(
|
||||
tmp_path / "p", "schema: canned-prompts/eval-rubric/v0.1\nname: empty\n"
|
||||
)
|
||||
with pytest.raises(cp.CannedPromptError, match="asserts nothing"):
|
||||
cp.validate_package(pkg)
|
||||
|
||||
|
||||
def test_unknown_render_check_is_rejected(tmp_path: Path) -> None:
|
||||
pkg = write_evaluated(
|
||||
tmp_path / "p",
|
||||
"schema: canned-prompts/eval-rubric/v0.1\nname: q\nrender:\n - matches: 'x.*'\n",
|
||||
)
|
||||
with pytest.raises(cp.CannedPromptError, match="unknown render check"):
|
||||
cp.validate_package(pkg)
|
||||
|
||||
|
||||
def test_eval_example_must_be_declared(tmp_path: Path) -> None:
|
||||
pkg = write_evaluated(
|
||||
tmp_path / "p",
|
||||
"schema: canned-prompts/eval-rubric/v0.1\nname: q\n"
|
||||
"example: examples/missing.yaml\nrender:\n - contains: x\n",
|
||||
)
|
||||
with pytest.raises(cp.CannedPromptError, match="not declared in the manifest"):
|
||||
cp.validate_package(pkg)
|
||||
|
||||
|
||||
def test_render_checks_evaluate(tmp_path: Path) -> None:
|
||||
resolution = cp.Resolution(values={"topic": "dogs"}, origins={"topic": "supplied"})
|
||||
checks = [{"contains": "dogs"}, {"contains": "cats"}, {"not_contains": "cats"}]
|
||||
outcomes = cp.run_render_checks("about dogs", resolution, checks)
|
||||
assert [ok for ok, _ in outcomes] == [True, False, True]
|
||||
|
||||
|
||||
def test_resolves_all_reports_unresolved_names() -> None:
|
||||
resolution = cp.Resolution(
|
||||
values={"a": 1}, origins={"a": "supplied", "b": "unresolved (no default)"}
|
||||
)
|
||||
outcomes = cp.run_render_checks("text", resolution, [{"resolves_all": True}])
|
||||
assert outcomes[0][0] is False
|
||||
assert "b" in outcomes[0][1]
|
||||
|
||||
|
||||
def test_typed_fixture_values_are_not_reparsed(tmp_path: Path) -> None:
|
||||
"""A YAML fixture carries real types; only CLI strings need parsing."""
|
||||
manifest = {"parameters": {"flag": {"type": "boolean", "default": False}}}
|
||||
assert cp.resolve_inputs(manifest, {"flag": True}).values["flag"] is True
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue