From 7779768058dbd778744a03fa4828d09eb81d046f Mon Sep 17 00:00:00 2001 From: tegwick Date: Mon, 28 Sep 2026 12:17:13 +0200 Subject: [PATCH] Reject aborted runs and incomplete classification evidence Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4 --- docs/TestDriverGeneralisationReview.md | 42 +++++ research/decisions/README.md | 1 + src/testdriver/classification.py | 64 ++++++- src/testdriver/evidence.py | 3 + src/testdriver/runner.py | 31 +++- tests/test_evidence_completeness.py | 157 ++++++++++++++++++ workplans/TD-WP-0003-generalise-and-settle.md | 16 ++ 7 files changed, 311 insertions(+), 3 deletions(-) create mode 100644 tests/test_evidence_completeness.py diff --git a/docs/TestDriverGeneralisationReview.md b/docs/TestDriverGeneralisationReview.md index 2728982..b24922d 100644 --- a/docs/TestDriverGeneralisationReview.md +++ b/docs/TestDriverGeneralisationReview.md @@ -144,3 +144,45 @@ independent observation/cleanup receipts and appropriate approved access. A successful historical receipt cannot substitute for those. The existing T01, T06 and T07 tasks retain the experiment blockers; this review creates no production implementation commitment or new work record. + +## Safety follow-up: aborted runs and incomplete evidence + +A post-closeout review reproduced two false-pass paths that the original suite +missed. A forbidden-surface exception after a passing prefix returned PASS even +though later claims had never been judged. Separately, deleting almost every +judgment and all state snapshots still yielded UNCHANGED/safe-to-accept. + +The runner now records `scheduled_steps` and `expected_judgments` from the +scenario before executing any action. Only claims attached to scheduled steps +are included; the existing one-step browser scenario remains intentionally +narrow. On a surface violation, every remaining scheduled claim and invariant +gets an explicit INCONCLUSIVE judgment and no fabricated observation. A prior +FAIL still dominates. An abort after the last assertion is also INCONCLUSIVE, +because completing the assertions does not mean the scheduled run completed. + +The classifier validates both packs against their input manifests. It requires +exactly one judgment for every expected assertion/step pair and exactly one S1 +realization, S2 realization check and nonempty S3 state snapshot for every +scheduled step, with the correct strata. Missing, duplicate or invalid verdict +records, unfinished runs, and missing identity/version fields prevent automatic +acceptance. The candidate must retain the baseline's step and judgment coverage. +Evidence failure returns AMBIGUOUS before any adaptation rule is considered. + +This adds two serialized EvidencePack fields. Historical packs without these +manifests remain historical artifacts but classify as AMBIGUOUS; rerun the +scenario to obtain evidence eligible for automatic acceptance. Do not infer a +manifest from surviving observations, since that would preserve the original +hole. These checks establish structural completeness, not authenticity of a +pack or correctness of arbitrary caller-supplied predicates. The existing +actor/observer separation and deterministic oracle remain required. + +Regression coverage in `tests/test_evidence_completeness.py` includes aborts at +each step, abort after the final assertion, preservation of an earlier failure, +individual missing strata/records, duplicate records, and identical truncation +of both packs. Complete ordinary runs and intentionally narrow scenarios retain +their existing behavior. The two defects are handled under the existing T08 +readiness review; the external T01/T06/T07 blockers remain unchanged. + +Validation after the safety fixes: **299 tests passed** (`python3 -m pytest -q`). +The new regression module had 55 failures and five passing controls before the +fix; `git diff --check` passes after it. diff --git a/research/decisions/README.md b/research/decisions/README.md index f29d59f..d060457 100644 --- a/research/decisions/README.md +++ b/research/decisions/README.md @@ -8,3 +8,4 @@ file is a pointer table, not a second source of truth. |---|---|---|---| | D-01 … D-07 | Stratified evidence and claim provenance as the basis for adaptation safety | `fef5213f-ce9b-44c2-b327-a0b0ba4b6270` | `docs/TestDriverClassificationDesign.md` | | TD-WP-0003 T04/T05/T08 | Python claims, concept removal and not-ready assessment | `d80734d1-01b8-4a98-8e72-c86f1806d572` | `docs/TestDriverGeneralisationReview.md` | +| TD-WP-0003 T08 safety follow-up | Run-input manifests and incomplete-run handling | `afb38ade-893b-48a2-b1b3-c55e49cc40ee` | `docs/TestDriverGeneralisationReview.md` | diff --git a/src/testdriver/classification.py b/src/testdriver/classification.py index 4623111..5f9c6af 100644 --- a/src/testdriver/classification.py +++ b/src/testdriver/classification.py @@ -135,7 +135,64 @@ def _claim_fingerprint(pack: Mapping[str, Any]) -> str: return json.dumps(sorted(pack.get("provenance_index", {}).items())) +def _has_complete_evidence(pack: Mapping[str, Any]) -> bool: + """Validate retained coverage against the run inputs, not surviving records. + + This checks structural completeness, not authenticity or predicate truth. + Legacy packs without a schedule/expected-judgment manifest cannot establish + completeness and must be re-run before automatic acceptance. + """ + try: + if not all(pack.get(key) for key in ( + "run_id", "scenario_id", "use_case_id", "sut_version", "finished_at", + )): + return False + steps = pack["scheduled_steps"] + if not isinstance(steps, list) or not steps or not all( + isinstance(step, str) and step for step in steps + ) or len(set(steps)) != len(steps): + return False + expected = pack["expected_judgments"] + verdicts = pack["verdicts"] + if not isinstance(expected, list) or not expected or not isinstance(verdicts, list): + return False + expected_keys = [(v["assertion_id"], v["step_id"]) for v in expected] + if any(not isinstance(assertion, str) or not assertion or step not in steps + for assertion, step in expected_keys): + return False + actual_keys = [(v["assertion_id"], v["step_id"]) for v in verdicts] + if (len(set(expected_keys)) != len(expected_keys) + or len(set(actual_keys)) != len(actual_keys) + or set(expected_keys) != set(actual_keys) + or any(v["verdict"] not in ("PASS", "FAIL", "INCONCLUSIVE") for v in verdicts)): + return False + observations = pack["observations"] + if not isinstance(observations, list): + return False + for kind, stratum in (("realization", "S1"), ("realization_check", "S2"), + ("state_snapshot", "S3")): + records = [o for o in observations if o["kind"] == kind] + observed_steps = [o["step_id"] for o in records] + if (len(observed_steps) != len(steps) or set(observed_steps) != set(steps) + or any(o["stratum"] != stratum + or not isinstance(o["data"], Mapping) or not o["data"] + for o in records)): + return False + if any(o["kind"] == "surface_violation" for o in observations): + return False + return isinstance(pack["provenance_index"], Mapping) + except (KeyError, TypeError, ValueError): + # Missing/malformed records are evidence failure, not a default pass. + return False + + def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Signals: + if not _has_complete_evidence(baseline) or not _has_complete_evidence(candidate): + return Signals( + surface_differs=False, realization_failed=False, postcondition_met=None, + verdicts_regressed=False, verdicts_inconclusive=False, claims_differ=False, + evidence_incomplete=True, + ) before, after = _verdict_map(baseline), _verdict_map(candidate) regressed = any( @@ -167,7 +224,10 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) - verdicts_regressed=regressed, verdicts_inconclusive=any(v == "INCONCLUSIVE" for v in after.values()), claims_differ=_claim_fingerprint(baseline) != _claim_fingerprint(candidate), - evidence_incomplete=not after or not candidate.get("sut_version"), + evidence_incomplete=( + not before.keys() <= after.keys() + or not set(baseline["scheduled_steps"]) <= set(candidate["scheduled_steps"]) + ), ) @@ -192,13 +252,13 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco change that happens to break a claim. """ signals = extract_signals(baseline, candidate) - failures = regressions(baseline, candidate) # 1. Evidence first. A conclusion drawn from incomplete evidence is worse # than no conclusion, in either direction. if signals.evidence_incomplete: return Outcome(Classification.AMBIGUOUS, "the run did not retain enough evidence to classify", signals) + failures = regressions(baseline, candidate) if signals.verdicts_inconclusive: return Outcome(Classification.AMBIGUOUS, "at least one assertion could not be judged from the evidence", diff --git a/src/testdriver/evidence.py b/src/testdriver/evidence.py index d2cc204..b7261f8 100644 --- a/src/testdriver/evidence.py +++ b/src/testdriver/evidence.py @@ -64,6 +64,9 @@ class EvidencePack: observations: list[Observation] = field(default_factory=list) verdicts: list[dict[str, Any]] = field(default_factory=list) provenance_index: dict[str, str] = field(default_factory=dict) + # Run inputs, captured before execution. A surviving prefix is not a manifest. + scheduled_steps: list[str] = field(default_factory=list) + expected_judgments: list[dict[str, str]] = field(default_factory=list) def record(self, observation: Observation) -> None: self.observations.append(observation) diff --git a/src/testdriver/runner.py b/src/testdriver/runner.py index c4feceb..614acd6 100644 --- a/src/testdriver/runner.py +++ b/src/testdriver/runner.py @@ -118,6 +118,15 @@ class Runner: scenario_id=scenario.id, use_case_id=scenario.use_case.id, sut_version=self._world.sut_version, + scheduled_steps=[step.id for step in scenario.steps], + expected_judgments=[ + {"assertion_id": assertion.id, "step_id": step.id} + for step in scenario.steps + for assertion in ( + *scenario.use_case.invariants, + *(c for c in scenario.use_case.claims if c.after_step == step.id), + ) + ], ) pack.provenance_index = { scenario.use_case.id: scenario.use_case.provenance.value, @@ -133,11 +142,12 @@ class Runner: judgments: list[Judgment] = [] scenario_sound = True + aborted = False claims_by_step: dict[str, list] = {} for claim in scenario.use_case.claims: claims_by_step.setdefault(claim.after_step, []).append(claim) - for step in scenario.steps: + for step_index, step in enumerate(scenario.steps): actor = self._world.cast[step.actor_id] # --- S1: how it was done ------------------------------------- @@ -151,6 +161,21 @@ class Runner: {"step": step.id, "action": step.action.describe(), "error": str(exc)}, step.id, ) + aborted = True + # Keep the scheduled assertion set, but never invent observations + # for steps we did not execute. Claims outside this scenario's + # schedule remain outside its scope (e.g. the one-step browser arm). + for skipped in scenario.steps[step_index:]: + for assertion in ( + *scenario.use_case.invariants, + *claims_by_step.get(skipped.id, ()), + ): + judgments.append(Judgment( + assertion.id, assertion.text, Verdict.INCONCLUSIVE, + skipped.id, + {"reason": f"run aborted at {step.id!r}; " + "this scheduled step was not executed"}, + )) break self._record( @@ -225,4 +250,8 @@ class Runner: pack.finished_at = datetime.now(timezone.utc).isoformat() result_verdict = overall(judgments) + # Even an assertion-free trailing step is part of the scheduled run. + # Preserve observed failures, but a passing prefix cannot certify an abort. + if aborted and result_verdict is Verdict.PASS: + result_verdict = Verdict.INCONCLUSIVE return RunResult(run_id, result_verdict, judgments, pack) diff --git a/tests/test_evidence_completeness.py b/tests/test_evidence_completeness.py new file mode 100644 index 0000000..05ca4e8 --- /dev/null +++ b/tests/test_evidence_completeness.py @@ -0,0 +1,157 @@ +"""A passing prefix or a truncated pack must never become an accepted run.""" +import json +from copy import deepcopy +from dataclasses import replace + +import pytest + +from scenarios.alice_bob_carol import build +from testdriver import Runner, Verdict +from testdriver.classification import Classification, classify + + +def run_reference(*mutations, abort_at=None, claims_only=False): + world, driver, observer, asset, oracle = build(*mutations) + steps = list(asset.scenario.steps) + if abort_at is not None: + steps[abort_at] = replace( + steps[abort_at], + action=replace(steps[abort_at].action, permitted_surfaces=frozenset({"browser"})), + ) + case = asset.scenario.use_case + if claims_only: + # Every claim is judged after grant; the final step has no assertions. + case = replace(case, invariants=(), claims=tuple( + c for c in case.claims if c.after_step == "s2-grant" + )) + asset.scenario = replace(asset.scenario, steps=tuple(steps), use_case=case) + return Runner(world, driver, observer, oracle).run(asset), asset.scenario + + +@pytest.mark.parametrize("abort_at", [0, 1, 2]) +def test_aborted_run_retains_unreached_claims_and_invariants(abort_at): + result, scenario = run_reference(abort_at=abort_at) + assert result.verdict is Verdict.INCONCLUSIVE + judgments = {(j.assertion_id, j.step_id): j for j in result.judgments} + expected = { + (a.id, step.id) for step in scenario.steps + for a in (*scenario.use_case.invariants, *( + c for c in scenario.use_case.claims if c.after_step == step.id + )) + } + assert judgments.keys() == expected + for step in scenario.steps[abort_at:]: + assert all(j.verdict is Verdict.INCONCLUSIVE + for j in result.judgments if j.step_id == step.id) + assert any(o.kind == "surface_violation" for o in result.evidence.observations) + # Do not fabricate successful steps or observations to fill the evidence gap. + assert len([o for o in result.evidence.observations + if o.kind == "state_snapshot"]) == abort_at + + +def test_abort_without_remaining_assertions_still_cannot_pass(): + result, _ = run_reference(abort_at=2, claims_only=True) + assert all(j.verdict is Verdict.PASS for j in result.judgments) + assert result.verdict is Verdict.INCONCLUSIVE + + +def test_abort_preserves_a_previously_observed_failure(): + result, _ = run_reference("M16", abort_at=2) + assert result.verdict is Verdict.FAIL + assert result.judgment("c-bob-cannot-write").verdict is Verdict.FAIL + assert result.judgment("c-bob-revoked").verdict is Verdict.INCONCLUSIVE + + +@pytest.fixture +def baseline(): + result, _ = run_reference() + return json.loads(result.evidence.to_json()) + + +def test_complete_pack_is_still_accepted(baseline): + assert classify(baseline, deepcopy(baseline)).classification is Classification.UNCHANGED + + +def test_manifest_comes_from_schedule_even_if_execution_aborts(): + baseline, _ = run_reference() + aborted, _ = run_reference(abort_at=1) + assert aborted.evidence.scheduled_steps == baseline.evidence.scheduled_steps + assert aborted.evidence.expected_judgments == baseline.evidence.expected_judgments + + +def test_missing_verdict_and_all_snapshots_reproduces_reported_hole(baseline): + candidate = deepcopy(baseline) + candidate["verdicts"] = candidate["verdicts"][:1] + candidate["observations"] = [o for o in candidate["observations"] + if o["kind"] != "state_snapshot"] + outcome = classify(baseline, candidate) + assert outcome.classification is Classification.AMBIGUOUS + assert not outcome.safe_to_accept + + +@pytest.mark.parametrize("side", ["baseline", "candidate", "both"]) +@pytest.mark.parametrize("damage", [ + "one-verdict", "all-verdicts", "one-snapshot", "all-snapshots", "empty-snapshot", + "s1-snapshot", "missing-realization", "missing-check", "wrong-step", + "duplicate-verdict", "duplicate-snapshot", "invalid-verdict", "missing-version", + "unfinished", "missing-manifest", "missing-observations", "missing-verdict-field", +]) +def test_incomplete_evidence_cannot_be_accepted(baseline, side, damage): + candidate = deepcopy(baseline) + + def corrupt(pack): + if damage == "one-verdict": + pack["verdicts"].pop() + elif damage == "all-verdicts": + pack["verdicts"] = [] + elif damage in ("one-snapshot", "missing-realization", "missing-check"): + kind = {"one-snapshot": "state_snapshot", "missing-realization": "realization", + "missing-check": "realization_check"}[damage] + pack["observations"].remove(next(o for o in pack["observations"] if o["kind"] == kind)) + elif damage == "all-snapshots": + pack["observations"] = [o for o in pack["observations"] if o["kind"] != "state_snapshot"] + elif damage in ("empty-snapshot", "s1-snapshot", "wrong-step", "duplicate-snapshot"): + snapshot = next(o for o in pack["observations"] if o["kind"] == "state_snapshot") + if damage == "empty-snapshot": + snapshot["data"] = {} + elif damage == "s1-snapshot": + snapshot["stratum"] = "S1" + elif damage == "wrong-step": + snapshot["step_id"] = "not-a-scheduled-step" + else: + pack["observations"].append(deepcopy(snapshot)) + elif damage == "duplicate-verdict": + pack["verdicts"].append(deepcopy(pack["verdicts"][0])) + elif damage == "invalid-verdict": + pack["verdicts"][0]["verdict"] = "UNKNOWN" + elif damage == "missing-version": + del pack["sut_version"] + elif damage == "unfinished": + pack["finished_at"] = None + elif damage == "missing-manifest": + pack.pop("scheduled_steps", None) + pack.pop("expected_judgments", None) + elif damage == "missing-observations": + del pack["observations"] + elif damage == "missing-verdict-field": + del pack["verdicts"] + + if side in ("baseline", "both"): + corrupt(baseline) + if side in ("candidate", "both"): + corrupt(candidate) + outcome = classify(baseline, candidate) + assert outcome.classification is Classification.AMBIGUOUS + assert not outcome.safe_to_accept + assert outcome.signals.evidence_incomplete + + +def test_truncating_the_manifest_too_cannot_hide_a_missing_baseline_step(baseline): + candidate = deepcopy(baseline) + missing = candidate.get("scheduled_steps", ["s3-revoke"])[-1] + candidate["scheduled_steps"] = ["s1-create", "s2-grant"] + candidate["expected_judgments"] = [v for v in candidate.get("expected_judgments", []) + if v["step_id"] != missing] + candidate["verdicts"] = [v for v in candidate["verdicts"] if v["step_id"] != missing] + candidate["observations"] = [o for o in candidate["observations"] if o["step_id"] != missing] + assert not classify(baseline, candidate).safe_to_accept diff --git a/workplans/TD-WP-0003-generalise-and-settle.md b/workplans/TD-WP-0003-generalise-and-settle.md index 1f19cd6..e979fc4 100644 --- a/workplans/TD-WP-0003-generalise-and-settle.md +++ b/workplans/TD-WP-0003-generalise-and-settle.md @@ -299,3 +299,19 @@ Then state a verdict, including the verdict "not yet, and here is what is missin A readiness review that cannot conclude *not ready* is not a review. **2026-09-28 closeout — done.** Published explicit real-system gates and a NOT READY verdict for autonomous application, including D-07 cost, provenance from real backlogs, ambiguity, engagement-specific false-adaptation harm, custody/timing/cleanup and economics. The audit-core contract remains intent plus fixture calibration, not a runnable production engagement. + + +**2026-09-28 safety follow-up — done.** Fixed two reproduced readiness defects +under this existing task. Forbidden-surface aborts now retain all unreached +scheduled claims/invariants as INCONCLUSIVE, prevent a passing-prefix run verdict, +and preserve failures already observed. Evidence packs record scheduled steps +and expected judgments before execution; classification rejects incomplete +baseline or candidate evidence, including missing S1/S2/S3 records, missing or +duplicate verdicts, and identical truncation of both packs. Historical packs +without the input manifest require a rerun before automatic acceptance. + +Validation: the new regression module produced 55 failures before the fix; all +**299 tests pass** after it, including existing classification matrices and +intentional partial scenarios. `git diff --check` is clean. Decision: +`afb38ade-893b-48a2-b1b3-c55e49cc40ee`. No new task or workplan was created; +T01/T06/T07 and the blocked workplan state are unchanged.