From 3ce772466b100a8f51c8b4be7ca88c9730194577 Mon Sep 17 00:00:00 2001 From: tegwick Date: Mon, 28 Sep 2026 15:57:45 +0200 Subject: [PATCH] Preserve oracle semantics in generated regression judgments Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4 --- crystallized/test_grant_access.py | 41 ++++++- docs/TestDriverEvidenceAndVariants.md | 19 ++++ research/decisions/README.md | 1 + src/testdriver/crystallization.py | 28 ++++- src/testdriver/oracles.py | 57 +++------- tests/test_crystallization.py | 1 + tests/test_generated_judgments.py | 106 ++++++++++++++++++ .../TD-WP-0004-scope-evidence-and-variants.md | 17 +++ 8 files changed, 224 insertions(+), 46 deletions(-) create mode 100644 tests/test_generated_judgments.py diff --git a/crystallized/test_grant_access.py b/crystallized/test_grant_access.py index c2be009..bae5bcb 100644 --- a/crystallized/test_grant_access.py +++ b/crystallized/test_grant_access.py @@ -7,7 +7,7 @@ ancestor maturity: T1 descendant : va-grant-crystallized (T5 Deterministic) frozen from : 4 identical realizations surface version: lab-0.2.0-baseline -generated : 2026-08-22 +generated : 2026-09-28 Why this file exists -------------------- @@ -29,6 +29,7 @@ is a finding. from __future__ import annotations +import unittest import urllib.error import urllib.parse import urllib.request @@ -74,6 +75,36 @@ def authenticated_target(base_url, path): return target, urllib.request.build_opener(_OriginRedirectHandler(origin)) + +def evaluate_predicate(predicate, snapshot): + """Shared deterministic judgment semantics, embeddable without the framework.""" + if not snapshot: + return "INCONCLUSIVE", "no observations were collected" + try: + satisfied = predicate(snapshot) + except KeyError as missing: + return "INCONCLUSIVE", f"required observation {missing} missing from snapshot" + except Exception as exc: + return "INCONCLUSIVE", f"predicate raised {type(exc).__name__}: {exc}" + if type(satisfied) is not bool: + return "INCONCLUSIVE", "predicate did not return a boolean" + return ("PASS" if satisfied else "FAIL"), None + + +def assert_predicates(predicates, snapshot): + """Map oracle outcomes to pytest: FAIL dominates; INCONCLUSIVE is explicit skip.""" + judgments = [(text, *evaluate_predicate(predicate, snapshot)) + for predicate, text in predicates] + failures = [text for text, verdict, _ in judgments if verdict == "FAIL"] + if failures: + raise AssertionError("; ".join(failures)) + unknown = [f"{text}: {reason}" for text, verdict, reason in judgments + if verdict == "INCONCLUSIVE"] + if unknown or not judgments: + raise unittest.SkipTest("INCONCLUSIVE: " + ("; ".join(unknown) or "no assertions")) + + + def _post(base_url: str, token: str, path: str, fields: dict) -> int: target, opener = authenticated_target(base_url, path) request = urllib.request.Request( @@ -104,6 +135,8 @@ def test_grant_access(crystallized_world): assert realize(base_url, token) < 400, "the frozen realization no longer works" snapshot = observe() - assert _bob_can_read(snapshot), 'Bob can read R after the grant' - assert _carol_cannot_read(snapshot), 'Carol can never read R' - assert _bob_cannot_write(snapshot), 'A READ grant does not let Bob write R' + assert_predicates([ + (_bob_can_read, 'Bob can read R after the grant'), + (_carol_cannot_read, 'Carol can never read R'), + (_bob_cannot_write, 'A READ grant does not let Bob write R'), + ], snapshot) diff --git a/docs/TestDriverEvidenceAndVariants.md b/docs/TestDriverEvidenceAndVariants.md index cea90a5..e34901a 100644 --- a/docs/TestDriverEvidenceAndVariants.md +++ b/docs/TestDriverEvidenceAndVariants.md @@ -92,3 +92,22 @@ This API deliberately does not generate new assertions, skip/reorder steps, schedule races or infer required evidence from a successful implementation run. Evidence/preflight decision: `aa6c3e31-614c-4030-990c-945acef08515`. + +## Generated regression outcomes + +Generated pytest modules embed the same predicate evaluator used by Oracle and +continue to import the original independent predicates. They require no runtime +test-driver machinery; the outcome adapter uses stdlib unittest.SkipTest. + +True means PASS and False means FAIL. Empty snapshots, missing observation keys, +predicate exceptions and non-boolean results mean INCONCLUSIVE. The module +evaluates all supplied assertions: any FAIL takes precedence over INCONCLUSIVE, +regardless of assertion order. With no failure but an inconclusive assertion, +pytest reports a skip labeled INCONCLUSIVE, not a pass or product failure. +Use `pytest -rs` to display those reasons. A zero pytest exit status alone is not +proof that every generated assertion was verified; inspect skipped outcomes. + +The checked-in descendant is regenerated with its original protected predicates +and frozen path. Execution-based tests compare generated and checked-in judgments +with Oracle, and a subprocess test verifies actual pytest pass/fail/skip reporting. +Decision: `c06c8752-80db-4458-9eaf-6321a0ca710c`. diff --git a/research/decisions/README.md b/research/decisions/README.md index 73a572a..801c8d9 100644 --- a/research/decisions/README.md +++ b/research/decisions/README.md @@ -15,3 +15,4 @@ file is a pointer table, not a second source of truth. | TD-WP-0003 T08 HTTP/surface follow-up | Origin-bound credentials and runner surface validation | `88490ee8-839d-40dc-affa-b00a1c68e3c5` | `docs/TestDriverGeneralisationReview.md` | | TD-WP-0004 | Evidence-backed scope, local receipts and intent-preserving variants | `f3b35dce-6047-4a95-8083-4d7032d4e99f` | `history/2026-09-28-121933-scope-intent-assessment.md` | | TD-WP-0004 evidence/preflight follow-up | Lossless observation types and schedule preflight | `aa6c3e31-614c-4030-990c-945acef08515` | `docs/TestDriverEvidenceAndVariants.md` | +| TD-WP-0004 generated-judgment follow-up | Shared oracle semantics in standalone regressions | `c06c8752-80db-4458-9eaf-6321a0ca710c` | `docs/TestDriverEvidenceAndVariants.md` | diff --git a/src/testdriver/crystallization.py b/src/testdriver/crystallization.py index 16a3e71..13790e5 100644 --- a/src/testdriver/crystallization.py +++ b/src/testdriver/crystallization.py @@ -20,12 +20,14 @@ from __future__ import annotations import json import inspect +import unittest from dataclasses import dataclass, field from datetime import datetime, timezone from typing import Any, Mapping, Sequence from .actions import SemanticAction from .classification import classify +from .oracles import evaluate_predicate from .drivers import Realization from .world import Actor from .http import OriginViolation, _origin, _OriginRedirectHandler, authenticated_target @@ -187,6 +189,19 @@ class CrystallizedDriver: return Realization(trajectory.surface_id, mechanics) +def assert_predicates(predicates, snapshot): + """Map oracle outcomes to pytest: FAIL dominates; INCONCLUSIVE is explicit skip.""" + judgments = [(text, *evaluate_predicate(predicate, snapshot)) + for predicate, text in predicates] + failures = [text for text, verdict, _ in judgments if verdict == "FAIL"] + if failures: + raise AssertionError("; ".join(failures)) + unknown = [f"{text}: {reason}" for text, verdict, reason in judgments + if verdict == "INCONCLUSIVE"] + if unknown or not judgments: + raise unittest.SkipTest("INCONCLUSIVE: " + ("; ".join(unknown) or "no assertions")) + + # --- code generation ------------------------------------------------------ _TEMPLATE = '''"""Crystallized regression test — generated, do not edit by hand. @@ -220,6 +235,7 @@ is a finding. from __future__ import annotations +import unittest import urllib.error import urllib.parse import urllib.request @@ -233,6 +249,9 @@ FIELDS = {fields!r} {http_helpers} +{judgment_helpers} + + def _post(base_url: str, token: str, path: str, fields: dict) -> int: target, opener = authenticated_target(base_url, path) request = urllib.request.Request( @@ -283,9 +302,9 @@ def generate_test_module( fields = {name: str(action_args[name]) for name in trajectory.fields if name in action_args} predicate_names = [c.predicate.__name__ for c in claims] - assertions = "\n".join( - f" assert {c.predicate.__name__}(snapshot), {c.text!r}" for c in claims - ) + assertions = " assert_predicates([\n" + "\n".join( + f" ({c.predicate.__name__}, {c.text!r})," for c in claims + ) + "\n ], snapshot)" return _TEMPLATE.format( ancestor_id=ancestor_id, ancestor_maturity=ancestor_maturity, @@ -300,6 +319,9 @@ def generate_test_module( action_name=trajectory.action_name, test_name=trajectory.action_name, assertions=assertions, + judgment_helpers="\n\n".join(inspect.getsource(obj) for obj in ( + evaluate_predicate, assert_predicates, + )), http_helpers="\n\n".join(inspect.getsource(obj) for obj in ( OriginViolation, _origin, _OriginRedirectHandler, authenticated_target, )), diff --git a/src/testdriver/oracles.py b/src/testdriver/oracles.py index 8fa7860..e86ee6e 100644 --- a/src/testdriver/oracles.py +++ b/src/testdriver/oracles.py @@ -55,45 +55,24 @@ class Oracle: snapshot: Mapping[str, Any], step_id: str | None, ) -> Judgment: - if not snapshot: - return Judgment( - assertion.id, - assertion.text, - Verdict.INCONCLUSIVE, - step_id, - {"reason": "no observations were collected"}, - ) - try: - satisfied = assertion.predicate(snapshot) - except KeyError as missing: - # The evidence needed to judge this assertion was not collected. - # That is an evidence failure, never a pass and never a fail. - return Judgment( - assertion.id, - assertion.text, - Verdict.INCONCLUSIVE, - step_id, - {"reason": f"required observation {missing} missing from snapshot"}, - ) - except Exception as exc: - return Judgment( - assertion.id, - assertion.text, - Verdict.INCONCLUSIVE, - step_id, - {"reason": f"predicate raised {type(exc).__name__}: {exc}"}, - ) - if type(satisfied) is not bool: - return Judgment( - assertion.id, assertion.text, Verdict.INCONCLUSIVE, step_id, - {"reason": "predicate did not return a boolean"}, - ) - return Judgment( - assertion.id, - assertion.text, - Verdict.PASS if satisfied else Verdict.FAIL, - step_id, - ) + verdict, reason = evaluate_predicate(assertion.predicate, snapshot) + return Judgment(assertion.id, assertion.text, Verdict(verdict), step_id, + {"reason": reason} if reason else {}) + + +def evaluate_predicate(predicate, snapshot): + """Shared deterministic judgment semantics, embeddable without the framework.""" + if not snapshot: + return "INCONCLUSIVE", "no observations were collected" + try: + satisfied = predicate(snapshot) + except KeyError as missing: + return "INCONCLUSIVE", f"required observation {missing} missing from snapshot" + except Exception as exc: + return "INCONCLUSIVE", f"predicate raised {type(exc).__name__}: {exc}" + if type(satisfied) is not bool: + return "INCONCLUSIVE", "predicate did not return a boolean" + return ("PASS" if satisfied else "FAIL"), None def overall(judgments: list[Judgment]) -> Verdict: diff --git a/tests/test_crystallization.py b/tests/test_crystallization.py index 0b9f5a1..8ed23e6 100644 --- a/tests/test_crystallization.py +++ b/tests/test_crystallization.py @@ -190,6 +190,7 @@ def test_the_generated_module_imports_no_agentic_machinery(): ] assert imports == [ "from __future__ import annotations", + "import unittest", "from scenarios.alice_bob_carol import _bob_can_read, " "_bob_cannot_write, _carol_cannot_read", ], imports diff --git a/tests/test_generated_judgments.py b/tests/test_generated_judgments.py new file mode 100644 index 0000000..e86efd7 --- /dev/null +++ b/tests/test_generated_judgments.py @@ -0,0 +1,106 @@ +"""Execute generated artifacts to check parity with deterministic oracle judgments.""" +from dataclasses import replace +import sys +import types +import unittest + +import pytest + +from scenarios.alice_bob_carol import USE_CASE +from testdriver import Oracle, Verdict +from testdriver.crystallization import Trajectory, generate_test_module +from testdriver.oracles import overall + + +def yes(snapshot): return True + +def no(snapshot): return False + +def unknown(snapshot): return 'unknown' + +def absent(snapshot): return None + +def numeric(snapshot): return 1 + +def missing(snapshot): return snapshot['missing'] + +def broken(snapshot): raise RuntimeError('unavailable') + + +class CannotCoerce: + def __bool__(self): + raise AssertionError('must not coerce') + + +def opaque(snapshot): return CannotCoerce() + + +def generated(predicates, monkeypatch): + module = types.ModuleType('_generated_claims_test') + for predicate in predicates: + setattr(module, predicate.__name__, predicate) + monkeypatch.setitem(sys.modules, module.__name__, module) + claims = [replace(USE_CASE.claims[0], id=f'claim-{i}', predicate=predicate) + for i, predicate in enumerate(predicates)] + source = generate_test_module( + trajectory=Trajectory('step', 'grant_access', 'browser', '/grant', ()), + action_args={}, ancestor_id='a', ancestor_maturity='T1', descendant_id='b', + runs=3, sut_version='test', claims=claims, claims_module=module.__name__) + namespace = {} + exec(compile(source, '', 'exec'), namespace) + namespace['realize'] = lambda *args: 200 + return namespace['test_grant_access'], claims + + +def outcome(test, snapshot): + try: + test(('unused', 'synthetic-token', lambda: snapshot)) + except unittest.SkipTest as exc: + assert 'INCONCLUSIVE' in str(exc) + return Verdict.INCONCLUSIVE + except AssertionError: + return Verdict.FAIL + return Verdict.PASS + + +@pytest.mark.parametrize('predicates', [(yes,), (no,), (unknown,), (absent,), + (numeric,), (missing,), (broken,), (opaque,), + (unknown, no), (no, unknown), (yes, unknown)]) +@pytest.mark.parametrize('snapshot', [{'observed': True}, {}]) +def test_generated_verdicts_match_oracle_including_failure_precedence(monkeypatch, predicates, snapshot): + test, claims = generated(predicates, monkeypatch) + expected = overall([Oracle().judge(c, snapshot, 'step') for c in claims]) + assert outcome(test, snapshot) is expected + + +@pytest.mark.parametrize('predicate', [yes, no, unknown, absent, numeric, missing, broken, opaque]) +def test_checked_in_descendant_preserves_the_same_semantics(monkeypatch, predicate): + import crystallized.test_grant_access as artifact + monkeypatch.setattr(artifact, 'realize', lambda *args: 200) + for name in ('_bob_can_read', '_carol_cannot_read', '_bob_cannot_write'): + monkeypatch.setattr(artifact, name, predicate) + claim = replace(USE_CASE.claims[0], predicate=predicate) + assert outcome(artifact.test_grant_access, {'observed': True}) is Oracle().judge( + claim, {'observed': True}, 'step').verdict + + +def test_pytest_reports_inconclusive_as_skipped_and_never_as_passed(tmp_path): + import subprocess + (tmp_path / 'local_predicates.py').write_text( + "def yes(snapshot): return True\ndef no(snapshot): return False\n" + "def unknown(snapshot): return 'unknown'\n") + (tmp_path / 'conftest.py').write_text( + "import pytest\n@pytest.fixture\ndef crystallized_world():\n" + " return ('unused', 'synthetic', lambda: {'observed': True})\n") + for predicate in (yes, no, unknown): + source = generate_test_module( + trajectory=Trajectory('step', 'grant_access', 'browser', '/grant', ()), + action_args={}, ancestor_id='a', ancestor_maturity='T1', descendant_id='b', + runs=3, sut_version='test', claims=[replace(USE_CASE.claims[0], predicate=predicate)], + claims_module='local_predicates') + (tmp_path / f'test_{predicate.__name__}.py').write_text(source + '\nrealize = lambda *args: 200\n') + completed = subprocess.run([sys.executable, '-m', 'pytest', '-q', '-rs', str(tmp_path)], + cwd=tmp_path, text=True, capture_output=True, timeout=30) + assert completed.returncode == 1, completed.stdout + completed.stderr + assert '1 failed, 1 passed, 1 skipped' in completed.stdout + assert 'INCONCLUSIVE' in completed.stdout diff --git a/workplans/TD-WP-0004-scope-evidence-and-variants.md b/workplans/TD-WP-0004-scope-evidence-and-variants.md index 6d25a26..7e3e1e7 100644 --- a/workplans/TD-WP-0004-scope-evidence-and-variants.md +++ b/workplans/TD-WP-0004-scope-evidence-and-variants.md @@ -147,3 +147,20 @@ Validation: 12 new regressions, of which 10 failed before the fixes; **131 focus tests passed**, and the full suite passed **440 tests in 169.07 seconds**. `git diff --check` is clean. Decision: `aa6c3e31-614c-4030-990c-945acef08515`. No new tasks/workplans; T05 remains waiting and this workplan remains blocked. + + +**2026-09-28 generated-judgment follow-up — done (T04).** Generated regression +modules now embed the actual predicate evaluator used by Oracle, retaining strict +boolean and missing/exception/invalid-result INCONCLUSIVE semantics. The adapter +evaluates all protected assertions so FAIL dominates any INCONCLUSIVE result; +otherwise an inconclusive result becomes a stdlib SkipTest explicitly labeled +INCONCLUSIVE in pytest. Callers must inspect skipped outcomes, not treat exit zero +as complete verification. The checked-in descendant was regenerated with its +original predicate set, frozen route and lineage. No runtime test-driver import +or additional dependency is introduced into generated artifacts. + +Validation: the initial 30 parity cases reproduced 24 failures and six controls; +all 31 final parity cases pass, including native pytest reporting in a subprocess. +The related subset passed 93 tests. Full suite: **471 passed in 167.82 seconds**. +`git diff --check` is clean. Decision: `c06c8752-80db-4458-9eaf-6321a0ca710c`. +No new task/workplan; T05 remains waiting and this workplan remains blocked.