Prevent passing run verdicts after failed scheduled realizations
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
parent
3ce772466b
commit
e3aac44c59
3 changed files with 36 additions and 2 deletions
|
|
@ -326,8 +326,8 @@ class Runner:
|
||||||
|
|
||||||
result_verdict = overall(judgments)
|
result_verdict = overall(judgments)
|
||||||
# Even an assertion-free trailing step is part of the scheduled run.
|
# Even an assertion-free trailing step is part of the scheduled run.
|
||||||
# Preserve observed failures, but a passing prefix cannot certify an abort.
|
# Preserve failures, but passing assertions cannot certify an unrealized step.
|
||||||
if aborted and result_verdict is Verdict.PASS:
|
if (aborted or not scenario_sound) and result_verdict is Verdict.PASS:
|
||||||
result_verdict = Verdict.INCONCLUSIVE
|
result_verdict = Verdict.INCONCLUSIVE
|
||||||
pack.run_verdict = result_verdict.value
|
pack.run_verdict = result_verdict.value
|
||||||
if evidence_store is not None:
|
if evidence_store is not None:
|
||||||
|
|
|
||||||
|
|
@ -155,3 +155,26 @@ def test_truncating_the_manifest_too_cannot_hide_a_missing_baseline_step(baselin
|
||||||
candidate["verdicts"] = [v for v in candidate["verdicts"] if v["step_id"] != missing]
|
candidate["verdicts"] = [v for v in candidate["verdicts"] if v["step_id"] != missing]
|
||||||
candidate["observations"] = [o for o in candidate["observations"] if o["step_id"] != missing]
|
candidate["observations"] = [o for o in candidate["observations"] if o["step_id"] != missing]
|
||||||
assert not classify(baseline, candidate).safe_to_accept
|
assert not classify(baseline, candidate).safe_to_accept
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize('mutation,expected', [(None, Verdict.INCONCLUSIVE), ('M17', Verdict.FAIL)])
|
||||||
|
def test_failed_trailing_realization_cannot_leave_a_passing_receipt(tmp_path, mutation, expected):
|
||||||
|
from testdriver import Step, SemanticAction
|
||||||
|
from testdriver.storage import EvidenceStore
|
||||||
|
world, driver, observer, asset, oracle = build(*([mutation] if mutation else []))
|
||||||
|
asset.scenario = replace(
|
||||||
|
asset.scenario,
|
||||||
|
use_case=replace(asset.scenario.use_case, invariants=()),
|
||||||
|
steps=(*asset.scenario.steps, Step('trailing-read', 'bob', SemanticAction(
|
||||||
|
'read_resource', {'resource_id': 'absent'}, frozenset({'api'})))),
|
||||||
|
)
|
||||||
|
store = EvidenceStore(tmp_path)
|
||||||
|
result = Runner(world, driver, observer, oracle).run(asset, evidence_store=store)
|
||||||
|
assert result.verdict is expected
|
||||||
|
assert store.load(result.run_id)['run_verdict'] == expected.value
|
||||||
|
assert any(o.kind == 'realization' and o.step_id == 'trailing-read' and o.data['raised']
|
||||||
|
for o in result.evidence.observations)
|
||||||
|
assert not any(j.step_id == 'trailing-read' for j in result.judgments)
|
||||||
|
if not mutation:
|
||||||
|
assert all(j.verdict is Verdict.PASS for j in result.judgments)
|
||||||
|
assert not classify(store.load(result.run_id), store.load(result.run_id)).safe_to_accept
|
||||||
|
|
|
||||||
|
|
@ -164,3 +164,14 @@ all 31 final parity cases pass, including native pytest reporting in a subproces
|
||||||
The related subset passed 93 tests. Full suite: **471 passed in 167.82 seconds**.
|
The related subset passed 93 tests. Full suite: **471 passed in 167.82 seconds**.
|
||||||
`git diff --check` is clean. Decision: `c06c8752-80db-4458-9eaf-6321a0ca710c`.
|
`git diff --check` is clean. Decision: `c06c8752-80db-4458-9eaf-6321a0ca710c`.
|
||||||
No new task/workplan; T05 remains waiting and this workplan remains blocked.
|
No new task/workplan; T05 remains waiting and this workplan remains blocked.
|
||||||
|
|
||||||
|
|
||||||
|
**2026-09-28 final run-verdict follow-up — done (T04).** A failed scheduled
|
||||||
|
realization now prevents a passing aggregate even when no claims follow it.
|
||||||
|
RunResult and persisted run_verdict become INCONCLUSIVE; previously observed
|
||||||
|
FAIL is preserved. Added two receipt-level regressions (one reproduced the false
|
||||||
|
PASS before the fix; the prior-failure control already passed). Focused evidence
|
||||||
|
suite: **62 passed**. Final full suite: **473 passed in 167.98 seconds**.
|
||||||
|
`git diff --check` is clean. No new task/workplan; pilot T05 stays waiting and
|
||||||
|
TD-WP-0004 stays blocked. The session closes with existing external prerequisites
|
||||||
|
explicitly retained, not promoted to completion.
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue