`evals/` was a reserved path holding unvalidated blobs: section 12 named the directory and gave an illustrative snippet, but nothing was specified, so no tool could act on an eval file. Every eval file must now declare a `schema`, and CPF defines exactly one — `canned-prompts/eval-rubric/v0.1`. Unrecognized schemas stay legal and are skipped rather than rejected, so the format gains something actionable without becoming an evaluation language, which remains a non-goal. The schema splits along the same seam as T01 and T03. Render checks (`contains`, `not_contains`, `resolves_all`) assert properties of the rendered prompt text, need no model, and are therefore run by the reference CLI. Output criteria describe a good result and are declared but not run, because judging them requires a model. That division is now the format's consistent answer to "deterministic locally, or not". An eval references a fixture already declared in the manifest's `examples` rather than carrying its own copy, so an example that is also an eval fixture stays honest — both break together. An eval declares assessment and must not record outcomes. Results are run evidence and live outside the immutable package, per INTENT.md and section 17. Spec: 12 rewritten with 12.1, 18 (rules 17-18), 21 (`eval` verb). Reference CLI: read_eval, validate_eval, load_example_values, run_render_checks, cmd_eval; a failed render check exits non-zero. Tests 42 -> 51. examples/pqrst-estimate/evals/quality.yaml is a real eval with four render checks and four output criteria, and it passes. Fixes a latent bug reaching a fixture exposed: coerce_value assumed every value was a command-line string, so a YAML fixture carrying a real type (include_rationale: true) crashed on .lower(). Typed values are now validated but not re-parsed. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Bjefh8NUiEiahN4JLwoSKM Assistant: claude-code Assistant-Model: opus Assistant-Process: 388925@bnt-lap001 Assistant-Session: 3507023f-e0fd-4a1e-9d90-a0d4217d1502
617 lines
18 KiB
Python
617 lines
18 KiB
Python
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
import canned_prompts as cp
|
|
|
|
|
|
@pytest.fixture()
|
|
def package(tmp_path: Path) -> Path:
|
|
pkg = tmp_path / "pkg"
|
|
pkg.mkdir()
|
|
(pkg / "prompt.yaml").write_text(
|
|
"""\
|
|
format: canned-prompt/v0.1
|
|
id: demo/hello
|
|
name: Hello
|
|
version: 1.0.0
|
|
summary: Say hello.
|
|
template: prompt.md
|
|
inputs:
|
|
- name: person
|
|
required: true
|
|
parameters:
|
|
tone:
|
|
type: enum
|
|
values: [warm, formal]
|
|
default: warm
|
|
""",
|
|
encoding="utf-8",
|
|
)
|
|
(pkg / "prompt.md").write_text(
|
|
"Say hello to {{ person }} in a {{ tone }} tone.\n", encoding="utf-8"
|
|
)
|
|
return pkg
|
|
|
|
|
|
def test_validate_and_render(package: Path) -> None:
|
|
manifest = cp.validate_package(package)
|
|
values = cp.resolve_values(manifest, {"person": "Ada"})
|
|
rendered = cp.render_template((package / "prompt.md").read_text(), values)
|
|
assert rendered == "Say hello to Ada in a warm tone.\n"
|
|
|
|
|
|
def test_missing_required_input_fails(package: Path) -> None:
|
|
manifest = cp.validate_package(package)
|
|
with pytest.raises(cp.CannedPromptError, match="missing required input"):
|
|
cp.resolve_values(manifest, {})
|
|
|
|
|
|
def test_undeclared_placeholder_fails(package: Path) -> None:
|
|
(package / "prompt.md").write_text("{{ missing }}\n", encoding="utf-8")
|
|
with pytest.raises(cp.CannedPromptError, match="undeclared placeholders"):
|
|
cp.validate_package(package)
|
|
|
|
|
|
def write_pkg(pkg: Path, manifest: str, template: str = "{{ greeting }}\n") -> Path:
|
|
pkg.mkdir(exist_ok=True)
|
|
(pkg / "prompt.yaml").write_text(manifest, encoding="utf-8")
|
|
(pkg / "prompt.md").write_text(template, encoding="utf-8")
|
|
return pkg
|
|
|
|
|
|
BASE = """\
|
|
format: canned-prompt/v0.1
|
|
id: demo/defaults
|
|
name: Defaults
|
|
version: 1.0.0
|
|
summary: Exercise input defaults.
|
|
template: prompt.md
|
|
"""
|
|
|
|
|
|
def test_static_default_fills_optional_input(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default: "hello there"
|
|
""",
|
|
)
|
|
manifest = cp.validate_package(pkg)
|
|
resolution = cp.resolve_inputs(manifest, {})
|
|
assert resolution.values["greeting"] == "hello there"
|
|
assert resolution.origins["greeting"] == "default"
|
|
assert cp.render_template("{{ greeting }}", resolution.values) == "hello there"
|
|
|
|
|
|
def test_supplied_value_overrides_static_default(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default: "hello there"
|
|
""",
|
|
)
|
|
manifest = cp.validate_package(pkg)
|
|
resolution = cp.resolve_inputs(manifest, {"greeting": "hi"})
|
|
assert resolution.values["greeting"] == "hi"
|
|
assert resolution.origins["greeting"] == "supplied"
|
|
|
|
|
|
def test_derived_default_uses_static_fallback(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
dependencies:
|
|
prompts:
|
|
- id: context/greeting
|
|
version: 1.0.0
|
|
requirement: generate
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default:
|
|
derive: context/greeting
|
|
value: "(none)"
|
|
""",
|
|
)
|
|
manifest = cp.validate_package(pkg)
|
|
resolution = cp.resolve_inputs(manifest, {})
|
|
assert resolution.values["greeting"] == "(none)"
|
|
assert resolution.origins["greeting"] == "fallback (not derived)"
|
|
assert resolution.underivable == []
|
|
|
|
|
|
def test_derived_default_without_fallback_is_underivable(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
dependencies:
|
|
prompts:
|
|
- id: context/greeting
|
|
version: 1.0.0
|
|
requirement: generate
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default:
|
|
derive: context/greeting
|
|
""",
|
|
)
|
|
manifest = cp.validate_package(pkg)
|
|
resolution = cp.resolve_inputs(manifest, {})
|
|
assert resolution.underivable == ["greeting"]
|
|
assert "greeting" not in resolution.values
|
|
with pytest.raises(cp.CannedPromptError, match="unresolved placeholder"):
|
|
cp.render_template("{{ greeting }}", resolution.values)
|
|
|
|
|
|
def test_inline_derive_is_valid(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default:
|
|
derive:
|
|
prompt: Produce a greeting suited to the audience.
|
|
value: "(none)"
|
|
""",
|
|
)
|
|
manifest = cp.validate_package(pkg)
|
|
assert cp.resolve_inputs(manifest, {}).values["greeting"] == "(none)"
|
|
|
|
|
|
def test_default_with_required_true_fails(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
inputs:
|
|
- name: greeting
|
|
required: true
|
|
default: "hello"
|
|
""",
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="required: true"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
def test_derive_reference_must_be_declared(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default:
|
|
derive: context/greeting
|
|
""",
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="not declared in dependencies"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
def test_inline_derive_cannot_also_reference(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default:
|
|
derive:
|
|
prompt: Produce a greeting.
|
|
id: context/greeting
|
|
""",
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="cannot also reference"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
MINIMAL = """\
|
|
format: canned-prompt/v0.1
|
|
id: practice/thing
|
|
name: Thing
|
|
version: 1.0.0
|
|
summary: A thing.
|
|
template: prompt.md
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default: hi
|
|
"""
|
|
|
|
|
|
def test_parse_reference() -> None:
|
|
assert cp.parse_reference("practice/thing") == (None, "practice/thing")
|
|
assert cp.parse_reference("house:practice/thing") == ("house", "practice/thing")
|
|
with pytest.raises(cp.CannedPromptError, match="malformed reference"):
|
|
cp.parse_reference("house:")
|
|
|
|
|
|
def test_registry_name_falls_back_to_basename(tmp_path: Path) -> None:
|
|
registry = tmp_path / "upstream"
|
|
registry.mkdir()
|
|
assert cp.registry_name(registry) == "upstream"
|
|
|
|
|
|
def test_registry_name_from_manifest(tmp_path: Path) -> None:
|
|
registry = tmp_path / "some-dir"
|
|
registry.mkdir()
|
|
(registry / "registry.yaml").write_text(
|
|
"format: canned-prompt-registry/v0.1\nname: house\n", encoding="utf-8"
|
|
)
|
|
assert cp.registry_name(registry) == "house"
|
|
|
|
|
|
def test_registry_manifest_rejects_bad_format(tmp_path: Path) -> None:
|
|
registry = tmp_path / "r"
|
|
registry.mkdir()
|
|
(registry / "registry.yaml").write_text(
|
|
"format: something-else\nname: house\n", encoding="utf-8"
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="unsupported registry format"):
|
|
cp.read_registry_manifest(registry)
|
|
|
|
|
|
def test_registry_manifest_rejects_bad_policy(tmp_path: Path) -> None:
|
|
registry = tmp_path / "r"
|
|
registry.mkdir()
|
|
(registry / "registry.yaml").write_text(
|
|
"format: canned-prompt-registry/v0.1\n"
|
|
"name: house\n"
|
|
"namespaces:\n practice:\n policy: maybe\n",
|
|
encoding="utf-8",
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="policy must be"):
|
|
cp.read_registry_manifest(registry)
|
|
|
|
|
|
def test_namespace_policy(tmp_path: Path) -> None:
|
|
registry = tmp_path / "r"
|
|
registry.mkdir()
|
|
(registry / "registry.yaml").write_text(
|
|
"format: canned-prompt-registry/v0.1\n"
|
|
"name: house\n"
|
|
"namespaces:\n practice:\n owner: Ada\n policy: closed\n",
|
|
encoding="utf-8",
|
|
)
|
|
assert cp.namespace_policy(registry, "practice/thing")[0] == "closed"
|
|
assert cp.namespace_policy(registry, "scratch/thing")[0] == "open"
|
|
|
|
|
|
def install_into(catalog: Path, registry: str) -> None:
|
|
"""Place a package in the catalog under a given registry name."""
|
|
dst = cp.catalog_package_path(catalog, registry, "practice/thing", "1.0.0")
|
|
dst.mkdir(parents=True)
|
|
(dst / "prompt.yaml").write_text(MINIMAL, encoding="utf-8")
|
|
(dst / "prompt.md").write_text("{{ greeting }}\n", encoding="utf-8")
|
|
|
|
|
|
def test_same_id_from_two_registries_coexists(tmp_path: Path) -> None:
|
|
catalog = tmp_path / "catalog"
|
|
install_into(catalog, "house")
|
|
install_into(catalog, "upstream")
|
|
assert cp.catalog_registries(catalog) == ["house", "upstream"]
|
|
|
|
package_dir, registry = cp.resolve_installed(catalog, "house:practice/thing", None)
|
|
assert registry == "house"
|
|
assert package_dir.is_dir()
|
|
|
|
|
|
def test_bare_id_in_two_registries_is_ambiguous(tmp_path: Path) -> None:
|
|
catalog = tmp_path / "catalog"
|
|
install_into(catalog, "house")
|
|
install_into(catalog, "upstream")
|
|
with pytest.raises(cp.CannedPromptError, match="more than one registry"):
|
|
cp.resolve_installed(catalog, "practice/thing", None)
|
|
|
|
|
|
def test_bare_id_in_one_registry_resolves(tmp_path: Path) -> None:
|
|
catalog = tmp_path / "catalog"
|
|
install_into(catalog, "house")
|
|
_, registry = cp.resolve_installed(catalog, "practice/thing", None)
|
|
assert registry == "house"
|
|
|
|
|
|
def test_legacy_catalog_layout_is_reported(tmp_path: Path) -> None:
|
|
catalog = tmp_path / "catalog"
|
|
legacy = catalog / "practice" / "thing" / "1.0.0"
|
|
legacy.mkdir(parents=True)
|
|
(legacy / "prompt.yaml").write_text(MINIMAL, encoding="utf-8")
|
|
(legacy / "prompt.md").write_text("{{ greeting }}\n", encoding="utf-8")
|
|
with pytest.raises(cp.CannedPromptError, match="pre-registry-scoped"):
|
|
cp.resolve_installed(catalog, "practice/thing", None)
|
|
|
|
|
|
# --- version selectors (§ 10.1) ---
|
|
|
|
AVAILABLE = ["2.1.0", "2.0.0", "1.5.0", "1.0.0"]
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"available,selector,expected",
|
|
[
|
|
(AVAILABLE, None, "2.1.0"),
|
|
(AVAILABLE, "any", "2.1.0"),
|
|
(AVAILABLE, "newest", "2.1.0"),
|
|
(AVAILABLE, "1.5.0", "1.5.0"),
|
|
(AVAILABLE, ">=2.0.0", "2.1.0"),
|
|
(AVAILABLE, "9.9.9", None),
|
|
(AVAILABLE, ">=9.9.9", None),
|
|
(["1.5.0", "1.0.0"], ">=1.2.0", "1.5.0"),
|
|
([], "newest", None),
|
|
],
|
|
)
|
|
def test_select_version(available, selector, expected) -> None:
|
|
assert cp.select_version(available, selector) == expected
|
|
|
|
|
|
@pytest.mark.parametrize("bad", ["^1.0.0", "~1.2", "1.x", ">=nope", "", "latest"])
|
|
def test_range_syntax_is_rejected(bad) -> None:
|
|
with pytest.raises(cp.CannedPromptError):
|
|
cp.validate_version_selector(bad, "dep")
|
|
|
|
|
|
# --- composition (§ 10.2) ---
|
|
|
|
FRAGMENT = """\
|
|
format: canned-prompt/v0.1
|
|
id: style/house
|
|
name: House Style
|
|
version: 1.0.0
|
|
summary: Shared style block.
|
|
type: fragment
|
|
template: prompt.md
|
|
parameters:
|
|
tone:
|
|
type: enum
|
|
values: [neutral, blunt]
|
|
default: neutral
|
|
"""
|
|
|
|
COMPOSER = """\
|
|
format: canned-prompt/v0.1
|
|
id: review/change
|
|
name: Change Review
|
|
version: 1.0.0
|
|
summary: Composes the shared style block.
|
|
template: prompt.md
|
|
dependencies:
|
|
prompts:
|
|
- id: style/house
|
|
version: 1.0.0
|
|
inputs:
|
|
- name: house_style
|
|
required: false
|
|
default:
|
|
include: style/house
|
|
parameters:
|
|
tone:
|
|
type: enum
|
|
values: [neutral, blunt]
|
|
default: blunt
|
|
"""
|
|
|
|
|
|
def place(catalog: Path, registry: str, package_id: str, version: str,
|
|
manifest: str, template: str) -> None:
|
|
dst = cp.catalog_package_path(catalog, registry, package_id, version)
|
|
dst.mkdir(parents=True)
|
|
(dst / "prompt.yaml").write_text(manifest, encoding="utf-8")
|
|
(dst / "prompt.md").write_text(template, encoding="utf-8")
|
|
|
|
|
|
@pytest.fixture()
|
|
def composed(tmp_path: Path) -> Path:
|
|
catalog = tmp_path / "catalog"
|
|
place(catalog, "local", "style/house", "1.0.0", FRAGMENT, "Tone is {{ tone }}.")
|
|
place(catalog, "local", "review/change", "1.0.0", COMPOSER, "S: {{ house_style }}")
|
|
return catalog
|
|
|
|
|
|
def test_include_inlines_rendered_template(composed: Path) -> None:
|
|
package_dir, _ = cp.resolve_installed(composed, "review/change", None)
|
|
manifest = cp.validate_package(package_dir)
|
|
resolution = cp.resolve_inputs(manifest, {}, composer=cp.CatalogComposer(composed))
|
|
assert resolution.values["house_style"] == "Tone is blunt."
|
|
assert resolution.origins["house_style"] == "included from style/house"
|
|
|
|
|
|
def test_include_inherits_outer_parameters(composed: Path) -> None:
|
|
"""The fragment's own default is neutral; the including package says blunt."""
|
|
package_dir, _ = cp.resolve_installed(composed, "review/change", None)
|
|
manifest = cp.validate_package(package_dir)
|
|
resolution = cp.resolve_inputs(
|
|
manifest, {"tone": "neutral"}, composer=cp.CatalogComposer(composed)
|
|
)
|
|
assert resolution.values["house_style"] == "Tone is neutral."
|
|
|
|
|
|
def test_include_without_composer_falls_back_to_unresolved(composed: Path) -> None:
|
|
package_dir, _ = cp.resolve_installed(composed, "review/change", None)
|
|
manifest = cp.validate_package(package_dir)
|
|
resolution = cp.resolve_inputs(manifest, {})
|
|
assert resolution.underivable == ["house_style"]
|
|
|
|
|
|
def test_inclusion_cycle_is_detected(tmp_path: Path) -> None:
|
|
catalog = tmp_path / "catalog"
|
|
for this, other in (("cyc/a", "cyc/b"), ("cyc/b", "cyc/a")):
|
|
manifest = (
|
|
"format: canned-prompt/v0.1\n"
|
|
f"id: {this}\nname: X\nversion: 1.0.0\nsummary: s\ntemplate: prompt.md\n"
|
|
f"dependencies:\n prompts:\n - id: {other}\n version: 1.0.0\n"
|
|
f"inputs:\n - name: other\n required: false\n"
|
|
f" default:\n include: {other}\n"
|
|
)
|
|
place(catalog, "local", this, "1.0.0", manifest, "{{ other }}")
|
|
package_dir, _ = cp.resolve_installed(catalog, "cyc/a", None)
|
|
manifest = cp.validate_package(package_dir)
|
|
with pytest.raises(cp.CannedPromptError, match="inclusion cycle"):
|
|
cp.resolve_inputs(manifest, {}, composer=cp.CatalogComposer(catalog))
|
|
|
|
|
|
def test_include_and_derive_together_is_rejected(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
dependencies:
|
|
prompts:
|
|
- id: style/house
|
|
version: 1.0.0
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default:
|
|
include: style/house
|
|
derive: style/house
|
|
""",
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="at most one of"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
def test_composed_dependency_must_declare_a_version(tmp_path: Path) -> None:
|
|
pkg = write_pkg(
|
|
tmp_path / "p",
|
|
BASE
|
|
+ """\
|
|
dependencies:
|
|
prompts:
|
|
- id: style/house
|
|
inputs:
|
|
- name: greeting
|
|
required: false
|
|
default:
|
|
include: style/house
|
|
""",
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="must declare a version"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
# --- evals (§ 12) ---
|
|
|
|
EVAL_BASE = """\
|
|
format: canned-prompt/v0.1
|
|
id: demo/evaluated
|
|
name: Evaluated
|
|
version: 1.0.0
|
|
summary: Exercise eval files.
|
|
template: prompt.md
|
|
examples:
|
|
- examples/basic.yaml
|
|
evals:
|
|
- evals/quality.yaml
|
|
"""
|
|
|
|
|
|
def write_evaluated(pkg: Path, eval_body: str, template: str = "Sum to 100%. {{ topic }}\n") -> Path:
|
|
pkg.mkdir(exist_ok=True)
|
|
(pkg / "prompt.yaml").write_text(
|
|
EVAL_BASE + "inputs:\n - name: topic\n required: false\n default: cats\n",
|
|
encoding="utf-8",
|
|
)
|
|
(pkg / "prompt.md").write_text(template, encoding="utf-8")
|
|
(pkg / "examples").mkdir(exist_ok=True)
|
|
(pkg / "examples" / "basic.yaml").write_text(
|
|
"name: basic\nvalues:\n topic: dogs\n", encoding="utf-8"
|
|
)
|
|
(pkg / "evals").mkdir(exist_ok=True)
|
|
(pkg / "evals" / "quality.yaml").write_text(eval_body, encoding="utf-8")
|
|
return pkg
|
|
|
|
|
|
RUBRIC = """\
|
|
schema: canned-prompts/eval-rubric/v0.1
|
|
name: quality
|
|
example: examples/basic.yaml
|
|
render:
|
|
- contains: "Sum to 100%"
|
|
- not_contains: "{{"
|
|
- resolves_all: true
|
|
output:
|
|
criteria:
|
|
- Answers the question.
|
|
"""
|
|
|
|
|
|
def test_valid_eval_passes_validation(tmp_path: Path) -> None:
|
|
pkg = write_evaluated(tmp_path / "p", RUBRIC)
|
|
assert cp.validate_package(pkg)["id"] == "demo/evaluated"
|
|
|
|
|
|
def test_eval_must_declare_a_schema(tmp_path: Path) -> None:
|
|
pkg = write_evaluated(tmp_path / "p", "name: quality\nrender: []\n")
|
|
with pytest.raises(cp.CannedPromptError, match="must declare a schema"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
def test_unknown_eval_schema_is_ignored(tmp_path: Path) -> None:
|
|
pkg = write_evaluated(tmp_path / "p", "schema: someone/else/v1\nwhatever: true\n")
|
|
assert cp.validate_package(pkg)["id"] == "demo/evaluated"
|
|
|
|
|
|
def test_eval_asserting_nothing_is_rejected(tmp_path: Path) -> None:
|
|
pkg = write_evaluated(
|
|
tmp_path / "p", "schema: canned-prompts/eval-rubric/v0.1\nname: empty\n"
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="asserts nothing"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
def test_unknown_render_check_is_rejected(tmp_path: Path) -> None:
|
|
pkg = write_evaluated(
|
|
tmp_path / "p",
|
|
"schema: canned-prompts/eval-rubric/v0.1\nname: q\nrender:\n - matches: 'x.*'\n",
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="unknown render check"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
def test_eval_example_must_be_declared(tmp_path: Path) -> None:
|
|
pkg = write_evaluated(
|
|
tmp_path / "p",
|
|
"schema: canned-prompts/eval-rubric/v0.1\nname: q\n"
|
|
"example: examples/missing.yaml\nrender:\n - contains: x\n",
|
|
)
|
|
with pytest.raises(cp.CannedPromptError, match="not declared in the manifest"):
|
|
cp.validate_package(pkg)
|
|
|
|
|
|
def test_render_checks_evaluate(tmp_path: Path) -> None:
|
|
resolution = cp.Resolution(values={"topic": "dogs"}, origins={"topic": "supplied"})
|
|
checks = [{"contains": "dogs"}, {"contains": "cats"}, {"not_contains": "cats"}]
|
|
outcomes = cp.run_render_checks("about dogs", resolution, checks)
|
|
assert [ok for ok, _ in outcomes] == [True, False, True]
|
|
|
|
|
|
def test_resolves_all_reports_unresolved_names() -> None:
|
|
resolution = cp.Resolution(
|
|
values={"a": 1}, origins={"a": "supplied", "b": "unresolved (no default)"}
|
|
)
|
|
outcomes = cp.run_render_checks("text", resolution, [{"resolves_all": True}])
|
|
assert outcomes[0][0] is False
|
|
assert "b" in outcomes[0][1]
|
|
|
|
|
|
def test_typed_fixture_values_are_not_reparsed(tmp_path: Path) -> None:
|
|
"""A YAML fixture carries real types; only CLI strings need parsing."""
|
|
manifest = {"parameters": {"flag": {"type": "boolean", "default": False}}}
|
|
assert cp.resolve_inputs(manifest, {"flag": True}).values["flag"] is True
|