Reject aborted runs and incomplete classification evidence

Assistant: codex
Assistant-Model: gpt-6-astra
Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
tegwick 2026-09-28 12:17:13 +02:00
parent 081fb8c73c
commit 7779768058
7 changed files with 311 additions and 3 deletions

View file

@ -135,7 +135,64 @@ def _claim_fingerprint(pack: Mapping[str, Any]) -> str:
return json.dumps(sorted(pack.get("provenance_index", {}).items()))
def _has_complete_evidence(pack: Mapping[str, Any]) -> bool:
"""Validate retained coverage against the run inputs, not surviving records.
This checks structural completeness, not authenticity or predicate truth.
Legacy packs without a schedule/expected-judgment manifest cannot establish
completeness and must be re-run before automatic acceptance.
"""
try:
if not all(pack.get(key) for key in (
"run_id", "scenario_id", "use_case_id", "sut_version", "finished_at",
)):
return False
steps = pack["scheduled_steps"]
if not isinstance(steps, list) or not steps or not all(
isinstance(step, str) and step for step in steps
) or len(set(steps)) != len(steps):
return False
expected = pack["expected_judgments"]
verdicts = pack["verdicts"]
if not isinstance(expected, list) or not expected or not isinstance(verdicts, list):
return False
expected_keys = [(v["assertion_id"], v["step_id"]) for v in expected]
if any(not isinstance(assertion, str) or not assertion or step not in steps
for assertion, step in expected_keys):
return False
actual_keys = [(v["assertion_id"], v["step_id"]) for v in verdicts]
if (len(set(expected_keys)) != len(expected_keys)
or len(set(actual_keys)) != len(actual_keys)
or set(expected_keys) != set(actual_keys)
or any(v["verdict"] not in ("PASS", "FAIL", "INCONCLUSIVE") for v in verdicts)):
return False
observations = pack["observations"]
if not isinstance(observations, list):
return False
for kind, stratum in (("realization", "S1"), ("realization_check", "S2"),
("state_snapshot", "S3")):
records = [o for o in observations if o["kind"] == kind]
observed_steps = [o["step_id"] for o in records]
if (len(observed_steps) != len(steps) or set(observed_steps) != set(steps)
or any(o["stratum"] != stratum
or not isinstance(o["data"], Mapping) or not o["data"]
for o in records)):
return False
if any(o["kind"] == "surface_violation" for o in observations):
return False
return isinstance(pack["provenance_index"], Mapping)
except (KeyError, TypeError, ValueError):
# Missing/malformed records are evidence failure, not a default pass.
return False
def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Signals:
if not _has_complete_evidence(baseline) or not _has_complete_evidence(candidate):
return Signals(
surface_differs=False, realization_failed=False, postcondition_met=None,
verdicts_regressed=False, verdicts_inconclusive=False, claims_differ=False,
evidence_incomplete=True,
)
before, after = _verdict_map(baseline), _verdict_map(candidate)
regressed = any(
@ -167,7 +224,10 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -
verdicts_regressed=regressed,
verdicts_inconclusive=any(v == "INCONCLUSIVE" for v in after.values()),
claims_differ=_claim_fingerprint(baseline) != _claim_fingerprint(candidate),
evidence_incomplete=not after or not candidate.get("sut_version"),
evidence_incomplete=(
not before.keys() <= after.keys()
or not set(baseline["scheduled_steps"]) <= set(candidate["scheduled_steps"])
),
)
@ -192,13 +252,13 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco
change that happens to break a claim.
"""
signals = extract_signals(baseline, candidate)
failures = regressions(baseline, candidate)
# 1. Evidence first. A conclusion drawn from incomplete evidence is worse
# than no conclusion, in either direction.
if signals.evidence_incomplete:
return Outcome(Classification.AMBIGUOUS,
"the run did not retain enough evidence to classify", signals)
failures = regressions(baseline, candidate)
if signals.verdicts_inconclusive:
return Outcome(Classification.AMBIGUOUS,
"at least one assertion could not be judged from the evidence",

View file

@ -64,6 +64,9 @@ class EvidencePack:
observations: list[Observation] = field(default_factory=list)
verdicts: list[dict[str, Any]] = field(default_factory=list)
provenance_index: dict[str, str] = field(default_factory=dict)
# Run inputs, captured before execution. A surviving prefix is not a manifest.
scheduled_steps: list[str] = field(default_factory=list)
expected_judgments: list[dict[str, str]] = field(default_factory=list)
def record(self, observation: Observation) -> None:
self.observations.append(observation)

View file

@ -118,6 +118,15 @@ class Runner:
scenario_id=scenario.id,
use_case_id=scenario.use_case.id,
sut_version=self._world.sut_version,
scheduled_steps=[step.id for step in scenario.steps],
expected_judgments=[
{"assertion_id": assertion.id, "step_id": step.id}
for step in scenario.steps
for assertion in (
*scenario.use_case.invariants,
*(c for c in scenario.use_case.claims if c.after_step == step.id),
)
],
)
pack.provenance_index = {
scenario.use_case.id: scenario.use_case.provenance.value,
@ -133,11 +142,12 @@ class Runner:
judgments: list[Judgment] = []
scenario_sound = True
aborted = False
claims_by_step: dict[str, list] = {}
for claim in scenario.use_case.claims:
claims_by_step.setdefault(claim.after_step, []).append(claim)
for step in scenario.steps:
for step_index, step in enumerate(scenario.steps):
actor = self._world.cast[step.actor_id]
# --- S1: how it was done -------------------------------------
@ -151,6 +161,21 @@ class Runner:
{"step": step.id, "action": step.action.describe(), "error": str(exc)},
step.id,
)
aborted = True
# Keep the scheduled assertion set, but never invent observations
# for steps we did not execute. Claims outside this scenario's
# schedule remain outside its scope (e.g. the one-step browser arm).
for skipped in scenario.steps[step_index:]:
for assertion in (
*scenario.use_case.invariants,
*claims_by_step.get(skipped.id, ()),
):
judgments.append(Judgment(
assertion.id, assertion.text, Verdict.INCONCLUSIVE,
skipped.id,
{"reason": f"run aborted at {step.id!r}; "
"this scheduled step was not executed"},
))
break
self._record(
@ -225,4 +250,8 @@ class Runner:
pack.finished_at = datetime.now(timezone.utc).isoformat()
result_verdict = overall(judgments)
# Even an assertion-free trailing step is part of the scheduled run.
# Preserve observed failures, but a passing prefix cannot certify an abort.
if aborted and result_verdict is Verdict.PASS:
result_verdict = Verdict.INCONCLUSIVE
return RunResult(run_id, result_verdict, judgments, pack)