diff --git a/src/testdriver/runner.py b/src/testdriver/runner.py index d2d30e2..b219d93 100644 --- a/src/testdriver/runner.py +++ b/src/testdriver/runner.py @@ -326,8 +326,8 @@ class Runner: result_verdict = overall(judgments) # Even an assertion-free trailing step is part of the scheduled run. - # Preserve observed failures, but a passing prefix cannot certify an abort. - if aborted and result_verdict is Verdict.PASS: + # Preserve failures, but passing assertions cannot certify an unrealized step. + if (aborted or not scenario_sound) and result_verdict is Verdict.PASS: result_verdict = Verdict.INCONCLUSIVE pack.run_verdict = result_verdict.value if evidence_store is not None: diff --git a/tests/test_evidence_completeness.py b/tests/test_evidence_completeness.py index 05ca4e8..0417bb9 100644 --- a/tests/test_evidence_completeness.py +++ b/tests/test_evidence_completeness.py @@ -155,3 +155,26 @@ def test_truncating_the_manifest_too_cannot_hide_a_missing_baseline_step(baselin candidate["verdicts"] = [v for v in candidate["verdicts"] if v["step_id"] != missing] candidate["observations"] = [o for o in candidate["observations"] if o["step_id"] != missing] assert not classify(baseline, candidate).safe_to_accept + + +@pytest.mark.parametrize('mutation,expected', [(None, Verdict.INCONCLUSIVE), ('M17', Verdict.FAIL)]) +def test_failed_trailing_realization_cannot_leave_a_passing_receipt(tmp_path, mutation, expected): + from testdriver import Step, SemanticAction + from testdriver.storage import EvidenceStore + world, driver, observer, asset, oracle = build(*([mutation] if mutation else [])) + asset.scenario = replace( + asset.scenario, + use_case=replace(asset.scenario.use_case, invariants=()), + steps=(*asset.scenario.steps, Step('trailing-read', 'bob', SemanticAction( + 'read_resource', {'resource_id': 'absent'}, frozenset({'api'})))), + ) + store = EvidenceStore(tmp_path) + result = Runner(world, driver, observer, oracle).run(asset, evidence_store=store) + assert result.verdict is expected + assert store.load(result.run_id)['run_verdict'] == expected.value + assert any(o.kind == 'realization' and o.step_id == 'trailing-read' and o.data['raised'] + for o in result.evidence.observations) + assert not any(j.step_id == 'trailing-read' for j in result.judgments) + if not mutation: + assert all(j.verdict is Verdict.PASS for j in result.judgments) + assert not classify(store.load(result.run_id), store.load(result.run_id)).safe_to_accept diff --git a/workplans/TD-WP-0004-scope-evidence-and-variants.md b/workplans/TD-WP-0004-scope-evidence-and-variants.md index 7e3e1e7..aabfee7 100644 --- a/workplans/TD-WP-0004-scope-evidence-and-variants.md +++ b/workplans/TD-WP-0004-scope-evidence-and-variants.md @@ -164,3 +164,14 @@ all 31 final parity cases pass, including native pytest reporting in a subproces The related subset passed 93 tests. Full suite: **471 passed in 167.82 seconds**. `git diff --check` is clean. Decision: `c06c8752-80db-4458-9eaf-6321a0ca710c`. No new task/workplan; T05 remains waiting and this workplan remains blocked. + + +**2026-09-28 final run-verdict follow-up — done (T04).** A failed scheduled +realization now prevents a passing aggregate even when no claims follow it. +RunResult and persisted run_verdict become INCONCLUSIVE; previously observed +FAIL is preserved. Added two receipt-level regressions (one reproduced the false +PASS before the fix; the prior-failure control already passed). Focused evidence +suite: **62 passed**. Final full suite: **473 passed in 167.98 seconds**. +`git diff --check` is clean. No new task/workplan; pilot T05 stays waiting and +TD-WP-0004 stays blocked. The session closes with existing external prerequisites +explicitly retained, not promoted to completion.