Preserve oracle semantics in generated regression judgments
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
parent
10077edb8c
commit
3ce772466b
8 changed files with 224 additions and 46 deletions
|
|
@ -7,7 +7,7 @@ ancestor maturity: T1
|
|||
descendant : va-grant-crystallized (T5 Deterministic)
|
||||
frozen from : 4 identical realizations
|
||||
surface version: lab-0.2.0-baseline
|
||||
generated : 2026-08-22
|
||||
generated : 2026-09-28
|
||||
|
||||
Why this file exists
|
||||
--------------------
|
||||
|
|
@ -29,6 +29,7 @@ is a finding.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
import unittest
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
|
|
@ -74,6 +75,36 @@ def authenticated_target(base_url, path):
|
|||
return target, urllib.request.build_opener(_OriginRedirectHandler(origin))
|
||||
|
||||
|
||||
|
||||
def evaluate_predicate(predicate, snapshot):
|
||||
"""Shared deterministic judgment semantics, embeddable without the framework."""
|
||||
if not snapshot:
|
||||
return "INCONCLUSIVE", "no observations were collected"
|
||||
try:
|
||||
satisfied = predicate(snapshot)
|
||||
except KeyError as missing:
|
||||
return "INCONCLUSIVE", f"required observation {missing} missing from snapshot"
|
||||
except Exception as exc:
|
||||
return "INCONCLUSIVE", f"predicate raised {type(exc).__name__}: {exc}"
|
||||
if type(satisfied) is not bool:
|
||||
return "INCONCLUSIVE", "predicate did not return a boolean"
|
||||
return ("PASS" if satisfied else "FAIL"), None
|
||||
|
||||
|
||||
def assert_predicates(predicates, snapshot):
|
||||
"""Map oracle outcomes to pytest: FAIL dominates; INCONCLUSIVE is explicit skip."""
|
||||
judgments = [(text, *evaluate_predicate(predicate, snapshot))
|
||||
for predicate, text in predicates]
|
||||
failures = [text for text, verdict, _ in judgments if verdict == "FAIL"]
|
||||
if failures:
|
||||
raise AssertionError("; ".join(failures))
|
||||
unknown = [f"{text}: {reason}" for text, verdict, reason in judgments
|
||||
if verdict == "INCONCLUSIVE"]
|
||||
if unknown or not judgments:
|
||||
raise unittest.SkipTest("INCONCLUSIVE: " + ("; ".join(unknown) or "no assertions"))
|
||||
|
||||
|
||||
|
||||
def _post(base_url: str, token: str, path: str, fields: dict) -> int:
|
||||
target, opener = authenticated_target(base_url, path)
|
||||
request = urllib.request.Request(
|
||||
|
|
@ -104,6 +135,8 @@ def test_grant_access(crystallized_world):
|
|||
assert realize(base_url, token) < 400, "the frozen realization no longer works"
|
||||
|
||||
snapshot = observe()
|
||||
assert _bob_can_read(snapshot), 'Bob can read R after the grant'
|
||||
assert _carol_cannot_read(snapshot), 'Carol can never read R'
|
||||
assert _bob_cannot_write(snapshot), 'A READ grant does not let Bob write R'
|
||||
assert_predicates([
|
||||
(_bob_can_read, 'Bob can read R after the grant'),
|
||||
(_carol_cannot_read, 'Carol can never read R'),
|
||||
(_bob_cannot_write, 'A READ grant does not let Bob write R'),
|
||||
], snapshot)
|
||||
|
|
|
|||
|
|
@ -92,3 +92,22 @@ This API deliberately does not generate new assertions, skip/reorder steps,
|
|||
schedule races or infer required evidence from a successful implementation run.
|
||||
|
||||
Evidence/preflight decision: `aa6c3e31-614c-4030-990c-945acef08515`.
|
||||
|
||||
## Generated regression outcomes
|
||||
|
||||
Generated pytest modules embed the same predicate evaluator used by Oracle and
|
||||
continue to import the original independent predicates. They require no runtime
|
||||
test-driver machinery; the outcome adapter uses stdlib unittest.SkipTest.
|
||||
|
||||
True means PASS and False means FAIL. Empty snapshots, missing observation keys,
|
||||
predicate exceptions and non-boolean results mean INCONCLUSIVE. The module
|
||||
evaluates all supplied assertions: any FAIL takes precedence over INCONCLUSIVE,
|
||||
regardless of assertion order. With no failure but an inconclusive assertion,
|
||||
pytest reports a skip labeled INCONCLUSIVE, not a pass or product failure.
|
||||
Use `pytest -rs` to display those reasons. A zero pytest exit status alone is not
|
||||
proof that every generated assertion was verified; inspect skipped outcomes.
|
||||
|
||||
The checked-in descendant is regenerated with its original protected predicates
|
||||
and frozen path. Execution-based tests compare generated and checked-in judgments
|
||||
with Oracle, and a subprocess test verifies actual pytest pass/fail/skip reporting.
|
||||
Decision: `c06c8752-80db-4458-9eaf-6321a0ca710c`.
|
||||
|
|
|
|||
|
|
@ -15,3 +15,4 @@ file is a pointer table, not a second source of truth.
|
|||
| TD-WP-0003 T08 HTTP/surface follow-up | Origin-bound credentials and runner surface validation | `88490ee8-839d-40dc-affa-b00a1c68e3c5` | `docs/TestDriverGeneralisationReview.md` |
|
||||
| TD-WP-0004 | Evidence-backed scope, local receipts and intent-preserving variants | `f3b35dce-6047-4a95-8083-4d7032d4e99f` | `history/2026-09-28-121933-scope-intent-assessment.md` |
|
||||
| TD-WP-0004 evidence/preflight follow-up | Lossless observation types and schedule preflight | `aa6c3e31-614c-4030-990c-945acef08515` | `docs/TestDriverEvidenceAndVariants.md` |
|
||||
| TD-WP-0004 generated-judgment follow-up | Shared oracle semantics in standalone regressions | `c06c8752-80db-4458-9eaf-6321a0ca710c` | `docs/TestDriverEvidenceAndVariants.md` |
|
||||
|
|
|
|||
|
|
@ -20,12 +20,14 @@ from __future__ import annotations
|
|||
|
||||
import json
|
||||
import inspect
|
||||
import unittest
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
from .actions import SemanticAction
|
||||
from .classification import classify
|
||||
from .oracles import evaluate_predicate
|
||||
from .drivers import Realization
|
||||
from .world import Actor
|
||||
from .http import OriginViolation, _origin, _OriginRedirectHandler, authenticated_target
|
||||
|
|
@ -187,6 +189,19 @@ class CrystallizedDriver:
|
|||
return Realization(trajectory.surface_id, mechanics)
|
||||
|
||||
|
||||
def assert_predicates(predicates, snapshot):
|
||||
"""Map oracle outcomes to pytest: FAIL dominates; INCONCLUSIVE is explicit skip."""
|
||||
judgments = [(text, *evaluate_predicate(predicate, snapshot))
|
||||
for predicate, text in predicates]
|
||||
failures = [text for text, verdict, _ in judgments if verdict == "FAIL"]
|
||||
if failures:
|
||||
raise AssertionError("; ".join(failures))
|
||||
unknown = [f"{text}: {reason}" for text, verdict, reason in judgments
|
||||
if verdict == "INCONCLUSIVE"]
|
||||
if unknown or not judgments:
|
||||
raise unittest.SkipTest("INCONCLUSIVE: " + ("; ".join(unknown) or "no assertions"))
|
||||
|
||||
|
||||
# --- code generation ------------------------------------------------------
|
||||
|
||||
_TEMPLATE = '''"""Crystallized regression test — generated, do not edit by hand.
|
||||
|
|
@ -220,6 +235,7 @@ is a finding.
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
import unittest
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
|
|
@ -233,6 +249,9 @@ FIELDS = {fields!r}
|
|||
{http_helpers}
|
||||
|
||||
|
||||
{judgment_helpers}
|
||||
|
||||
|
||||
def _post(base_url: str, token: str, path: str, fields: dict) -> int:
|
||||
target, opener = authenticated_target(base_url, path)
|
||||
request = urllib.request.Request(
|
||||
|
|
@ -283,9 +302,9 @@ def generate_test_module(
|
|||
fields = {name: str(action_args[name]) for name in trajectory.fields
|
||||
if name in action_args}
|
||||
predicate_names = [c.predicate.__name__ for c in claims]
|
||||
assertions = "\n".join(
|
||||
f" assert {c.predicate.__name__}(snapshot), {c.text!r}" for c in claims
|
||||
)
|
||||
assertions = " assert_predicates([\n" + "\n".join(
|
||||
f" ({c.predicate.__name__}, {c.text!r})," for c in claims
|
||||
) + "\n ], snapshot)"
|
||||
return _TEMPLATE.format(
|
||||
ancestor_id=ancestor_id,
|
||||
ancestor_maturity=ancestor_maturity,
|
||||
|
|
@ -300,6 +319,9 @@ def generate_test_module(
|
|||
action_name=trajectory.action_name,
|
||||
test_name=trajectory.action_name,
|
||||
assertions=assertions,
|
||||
judgment_helpers="\n\n".join(inspect.getsource(obj) for obj in (
|
||||
evaluate_predicate, assert_predicates,
|
||||
)),
|
||||
http_helpers="\n\n".join(inspect.getsource(obj) for obj in (
|
||||
OriginViolation, _origin, _OriginRedirectHandler, authenticated_target,
|
||||
)),
|
||||
|
|
|
|||
|
|
@ -55,45 +55,24 @@ class Oracle:
|
|||
snapshot: Mapping[str, Any],
|
||||
step_id: str | None,
|
||||
) -> Judgment:
|
||||
verdict, reason = evaluate_predicate(assertion.predicate, snapshot)
|
||||
return Judgment(assertion.id, assertion.text, Verdict(verdict), step_id,
|
||||
{"reason": reason} if reason else {})
|
||||
|
||||
|
||||
def evaluate_predicate(predicate, snapshot):
|
||||
"""Shared deterministic judgment semantics, embeddable without the framework."""
|
||||
if not snapshot:
|
||||
return Judgment(
|
||||
assertion.id,
|
||||
assertion.text,
|
||||
Verdict.INCONCLUSIVE,
|
||||
step_id,
|
||||
{"reason": "no observations were collected"},
|
||||
)
|
||||
return "INCONCLUSIVE", "no observations were collected"
|
||||
try:
|
||||
satisfied = assertion.predicate(snapshot)
|
||||
satisfied = predicate(snapshot)
|
||||
except KeyError as missing:
|
||||
# The evidence needed to judge this assertion was not collected.
|
||||
# That is an evidence failure, never a pass and never a fail.
|
||||
return Judgment(
|
||||
assertion.id,
|
||||
assertion.text,
|
||||
Verdict.INCONCLUSIVE,
|
||||
step_id,
|
||||
{"reason": f"required observation {missing} missing from snapshot"},
|
||||
)
|
||||
return "INCONCLUSIVE", f"required observation {missing} missing from snapshot"
|
||||
except Exception as exc:
|
||||
return Judgment(
|
||||
assertion.id,
|
||||
assertion.text,
|
||||
Verdict.INCONCLUSIVE,
|
||||
step_id,
|
||||
{"reason": f"predicate raised {type(exc).__name__}: {exc}"},
|
||||
)
|
||||
return "INCONCLUSIVE", f"predicate raised {type(exc).__name__}: {exc}"
|
||||
if type(satisfied) is not bool:
|
||||
return Judgment(
|
||||
assertion.id, assertion.text, Verdict.INCONCLUSIVE, step_id,
|
||||
{"reason": "predicate did not return a boolean"},
|
||||
)
|
||||
return Judgment(
|
||||
assertion.id,
|
||||
assertion.text,
|
||||
Verdict.PASS if satisfied else Verdict.FAIL,
|
||||
step_id,
|
||||
)
|
||||
return "INCONCLUSIVE", "predicate did not return a boolean"
|
||||
return ("PASS" if satisfied else "FAIL"), None
|
||||
|
||||
|
||||
def overall(judgments: list[Judgment]) -> Verdict:
|
||||
|
|
|
|||
|
|
@ -190,6 +190,7 @@ def test_the_generated_module_imports_no_agentic_machinery():
|
|||
]
|
||||
assert imports == [
|
||||
"from __future__ import annotations",
|
||||
"import unittest",
|
||||
"from scenarios.alice_bob_carol import _bob_can_read, "
|
||||
"_bob_cannot_write, _carol_cannot_read",
|
||||
], imports
|
||||
|
|
|
|||
106
tests/test_generated_judgments.py
Normal file
106
tests/test_generated_judgments.py
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
"""Execute generated artifacts to check parity with deterministic oracle judgments."""
|
||||
from dataclasses import replace
|
||||
import sys
|
||||
import types
|
||||
import unittest
|
||||
|
||||
import pytest
|
||||
|
||||
from scenarios.alice_bob_carol import USE_CASE
|
||||
from testdriver import Oracle, Verdict
|
||||
from testdriver.crystallization import Trajectory, generate_test_module
|
||||
from testdriver.oracles import overall
|
||||
|
||||
|
||||
def yes(snapshot): return True
|
||||
|
||||
def no(snapshot): return False
|
||||
|
||||
def unknown(snapshot): return 'unknown'
|
||||
|
||||
def absent(snapshot): return None
|
||||
|
||||
def numeric(snapshot): return 1
|
||||
|
||||
def missing(snapshot): return snapshot['missing']
|
||||
|
||||
def broken(snapshot): raise RuntimeError('unavailable')
|
||||
|
||||
|
||||
class CannotCoerce:
|
||||
def __bool__(self):
|
||||
raise AssertionError('must not coerce')
|
||||
|
||||
|
||||
def opaque(snapshot): return CannotCoerce()
|
||||
|
||||
|
||||
def generated(predicates, monkeypatch):
|
||||
module = types.ModuleType('_generated_claims_test')
|
||||
for predicate in predicates:
|
||||
setattr(module, predicate.__name__, predicate)
|
||||
monkeypatch.setitem(sys.modules, module.__name__, module)
|
||||
claims = [replace(USE_CASE.claims[0], id=f'claim-{i}', predicate=predicate)
|
||||
for i, predicate in enumerate(predicates)]
|
||||
source = generate_test_module(
|
||||
trajectory=Trajectory('step', 'grant_access', 'browser', '/grant', ()),
|
||||
action_args={}, ancestor_id='a', ancestor_maturity='T1', descendant_id='b',
|
||||
runs=3, sut_version='test', claims=claims, claims_module=module.__name__)
|
||||
namespace = {}
|
||||
exec(compile(source, '<generated>', 'exec'), namespace)
|
||||
namespace['realize'] = lambda *args: 200
|
||||
return namespace['test_grant_access'], claims
|
||||
|
||||
|
||||
def outcome(test, snapshot):
|
||||
try:
|
||||
test(('unused', 'synthetic-token', lambda: snapshot))
|
||||
except unittest.SkipTest as exc:
|
||||
assert 'INCONCLUSIVE' in str(exc)
|
||||
return Verdict.INCONCLUSIVE
|
||||
except AssertionError:
|
||||
return Verdict.FAIL
|
||||
return Verdict.PASS
|
||||
|
||||
|
||||
@pytest.mark.parametrize('predicates', [(yes,), (no,), (unknown,), (absent,),
|
||||
(numeric,), (missing,), (broken,), (opaque,),
|
||||
(unknown, no), (no, unknown), (yes, unknown)])
|
||||
@pytest.mark.parametrize('snapshot', [{'observed': True}, {}])
|
||||
def test_generated_verdicts_match_oracle_including_failure_precedence(monkeypatch, predicates, snapshot):
|
||||
test, claims = generated(predicates, monkeypatch)
|
||||
expected = overall([Oracle().judge(c, snapshot, 'step') for c in claims])
|
||||
assert outcome(test, snapshot) is expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize('predicate', [yes, no, unknown, absent, numeric, missing, broken, opaque])
|
||||
def test_checked_in_descendant_preserves_the_same_semantics(monkeypatch, predicate):
|
||||
import crystallized.test_grant_access as artifact
|
||||
monkeypatch.setattr(artifact, 'realize', lambda *args: 200)
|
||||
for name in ('_bob_can_read', '_carol_cannot_read', '_bob_cannot_write'):
|
||||
monkeypatch.setattr(artifact, name, predicate)
|
||||
claim = replace(USE_CASE.claims[0], predicate=predicate)
|
||||
assert outcome(artifact.test_grant_access, {'observed': True}) is Oracle().judge(
|
||||
claim, {'observed': True}, 'step').verdict
|
||||
|
||||
|
||||
def test_pytest_reports_inconclusive_as_skipped_and_never_as_passed(tmp_path):
|
||||
import subprocess
|
||||
(tmp_path / 'local_predicates.py').write_text(
|
||||
"def yes(snapshot): return True\ndef no(snapshot): return False\n"
|
||||
"def unknown(snapshot): return 'unknown'\n")
|
||||
(tmp_path / 'conftest.py').write_text(
|
||||
"import pytest\n@pytest.fixture\ndef crystallized_world():\n"
|
||||
" return ('unused', 'synthetic', lambda: {'observed': True})\n")
|
||||
for predicate in (yes, no, unknown):
|
||||
source = generate_test_module(
|
||||
trajectory=Trajectory('step', 'grant_access', 'browser', '/grant', ()),
|
||||
action_args={}, ancestor_id='a', ancestor_maturity='T1', descendant_id='b',
|
||||
runs=3, sut_version='test', claims=[replace(USE_CASE.claims[0], predicate=predicate)],
|
||||
claims_module='local_predicates')
|
||||
(tmp_path / f'test_{predicate.__name__}.py').write_text(source + '\nrealize = lambda *args: 200\n')
|
||||
completed = subprocess.run([sys.executable, '-m', 'pytest', '-q', '-rs', str(tmp_path)],
|
||||
cwd=tmp_path, text=True, capture_output=True, timeout=30)
|
||||
assert completed.returncode == 1, completed.stdout + completed.stderr
|
||||
assert '1 failed, 1 passed, 1 skipped' in completed.stdout
|
||||
assert 'INCONCLUSIVE' in completed.stdout
|
||||
|
|
@ -147,3 +147,20 @@ Validation: 12 new regressions, of which 10 failed before the fixes; **131 focus
|
|||
tests passed**, and the full suite passed **440 tests in 169.07 seconds**.
|
||||
`git diff --check` is clean. Decision: `aa6c3e31-614c-4030-990c-945acef08515`.
|
||||
No new tasks/workplans; T05 remains waiting and this workplan remains blocked.
|
||||
|
||||
|
||||
**2026-09-28 generated-judgment follow-up — done (T04).** Generated regression
|
||||
modules now embed the actual predicate evaluator used by Oracle, retaining strict
|
||||
boolean and missing/exception/invalid-result INCONCLUSIVE semantics. The adapter
|
||||
evaluates all protected assertions so FAIL dominates any INCONCLUSIVE result;
|
||||
otherwise an inconclusive result becomes a stdlib SkipTest explicitly labeled
|
||||
INCONCLUSIVE in pytest. Callers must inspect skipped outcomes, not treat exit zero
|
||||
as complete verification. The checked-in descendant was regenerated with its
|
||||
original predicate set, frozen route and lineage. No runtime test-driver import
|
||||
or additional dependency is introduced into generated artifacts.
|
||||
|
||||
Validation: the initial 30 parity cases reproduced 24 failures and six controls;
|
||||
all 31 final parity cases pass, including native pytest reporting in a subprocess.
|
||||
The related subset passed 93 tests. Full suite: **471 passed in 167.82 seconds**.
|
||||
`git diff --check` is clean. Decision: `c06c8752-80db-4458-9eaf-6321a0ca710c`.
|
||||
No new task/workplan; T05 remains waiting and this workplan remains blocked.
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue