Prevent passing run verdicts after failed scheduled realizations

Assistant: codex
Assistant-Model: gpt-6-astra
Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
tegwick 2026-09-28 16:28:05 +02:00
parent 3ce772466b
commit e3aac44c59
3 changed files with 36 additions and 2 deletions

View file

@ -326,8 +326,8 @@ class Runner:
result_verdict = overall(judgments)
# Even an assertion-free trailing step is part of the scheduled run.
# Preserve observed failures, but a passing prefix cannot certify an abort.
if aborted and result_verdict is Verdict.PASS:
# Preserve failures, but passing assertions cannot certify an unrealized step.
if (aborted or not scenario_sound) and result_verdict is Verdict.PASS:
result_verdict = Verdict.INCONCLUSIVE
pack.run_verdict = result_verdict.value
if evidence_store is not None:

View file

@ -155,3 +155,26 @@ def test_truncating_the_manifest_too_cannot_hide_a_missing_baseline_step(baselin
candidate["verdicts"] = [v for v in candidate["verdicts"] if v["step_id"] != missing]
candidate["observations"] = [o for o in candidate["observations"] if o["step_id"] != missing]
assert not classify(baseline, candidate).safe_to_accept
@pytest.mark.parametrize('mutation,expected', [(None, Verdict.INCONCLUSIVE), ('M17', Verdict.FAIL)])
def test_failed_trailing_realization_cannot_leave_a_passing_receipt(tmp_path, mutation, expected):
from testdriver import Step, SemanticAction
from testdriver.storage import EvidenceStore
world, driver, observer, asset, oracle = build(*([mutation] if mutation else []))
asset.scenario = replace(
asset.scenario,
use_case=replace(asset.scenario.use_case, invariants=()),
steps=(*asset.scenario.steps, Step('trailing-read', 'bob', SemanticAction(
'read_resource', {'resource_id': 'absent'}, frozenset({'api'})))),
)
store = EvidenceStore(tmp_path)
result = Runner(world, driver, observer, oracle).run(asset, evidence_store=store)
assert result.verdict is expected
assert store.load(result.run_id)['run_verdict'] == expected.value
assert any(o.kind == 'realization' and o.step_id == 'trailing-read' and o.data['raised']
for o in result.evidence.observations)
assert not any(j.step_id == 'trailing-read' for j in result.judgments)
if not mutation:
assert all(j.verdict is Verdict.PASS for j in result.judgments)
assert not classify(store.load(result.run_id), store.load(result.run_id)).safe_to_accept

View file

@ -164,3 +164,14 @@ all 31 final parity cases pass, including native pytest reporting in a subproces
The related subset passed 93 tests. Full suite: **471 passed in 167.82 seconds**.
`git diff --check` is clean. Decision: `c06c8752-80db-4458-9eaf-6321a0ca710c`.
No new task/workplan; T05 remains waiting and this workplan remains blocked.
**2026-09-28 final run-verdict follow-up — done (T04).** A failed scheduled
realization now prevents a passing aggregate even when no claims follow it.
RunResult and persisted run_verdict become INCONCLUSIVE; previously observed
FAIL is preserved. Added two receipt-level regressions (one reproduced the false
PASS before the fix; the prior-failure control already passed). Focused evidence
suite: **62 passed**. Final full suite: **473 passed in 167.98 seconds**.
`git diff --check` is clean. No new task/workplan; pilot T05 stays waiting and
TD-WP-0004 stays blocked. The session closes with existing external prerequisites
explicitly retained, not promoted to completion.