Reject aborted runs and incomplete classification evidence
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
parent
081fb8c73c
commit
7779768058
7 changed files with 311 additions and 3 deletions
|
|
@ -135,7 +135,64 @@ def _claim_fingerprint(pack: Mapping[str, Any]) -> str:
|
|||
return json.dumps(sorted(pack.get("provenance_index", {}).items()))
|
||||
|
||||
|
||||
def _has_complete_evidence(pack: Mapping[str, Any]) -> bool:
|
||||
"""Validate retained coverage against the run inputs, not surviving records.
|
||||
|
||||
This checks structural completeness, not authenticity or predicate truth.
|
||||
Legacy packs without a schedule/expected-judgment manifest cannot establish
|
||||
completeness and must be re-run before automatic acceptance.
|
||||
"""
|
||||
try:
|
||||
if not all(pack.get(key) for key in (
|
||||
"run_id", "scenario_id", "use_case_id", "sut_version", "finished_at",
|
||||
)):
|
||||
return False
|
||||
steps = pack["scheduled_steps"]
|
||||
if not isinstance(steps, list) or not steps or not all(
|
||||
isinstance(step, str) and step for step in steps
|
||||
) or len(set(steps)) != len(steps):
|
||||
return False
|
||||
expected = pack["expected_judgments"]
|
||||
verdicts = pack["verdicts"]
|
||||
if not isinstance(expected, list) or not expected or not isinstance(verdicts, list):
|
||||
return False
|
||||
expected_keys = [(v["assertion_id"], v["step_id"]) for v in expected]
|
||||
if any(not isinstance(assertion, str) or not assertion or step not in steps
|
||||
for assertion, step in expected_keys):
|
||||
return False
|
||||
actual_keys = [(v["assertion_id"], v["step_id"]) for v in verdicts]
|
||||
if (len(set(expected_keys)) != len(expected_keys)
|
||||
or len(set(actual_keys)) != len(actual_keys)
|
||||
or set(expected_keys) != set(actual_keys)
|
||||
or any(v["verdict"] not in ("PASS", "FAIL", "INCONCLUSIVE") for v in verdicts)):
|
||||
return False
|
||||
observations = pack["observations"]
|
||||
if not isinstance(observations, list):
|
||||
return False
|
||||
for kind, stratum in (("realization", "S1"), ("realization_check", "S2"),
|
||||
("state_snapshot", "S3")):
|
||||
records = [o for o in observations if o["kind"] == kind]
|
||||
observed_steps = [o["step_id"] for o in records]
|
||||
if (len(observed_steps) != len(steps) or set(observed_steps) != set(steps)
|
||||
or any(o["stratum"] != stratum
|
||||
or not isinstance(o["data"], Mapping) or not o["data"]
|
||||
for o in records)):
|
||||
return False
|
||||
if any(o["kind"] == "surface_violation" for o in observations):
|
||||
return False
|
||||
return isinstance(pack["provenance_index"], Mapping)
|
||||
except (KeyError, TypeError, ValueError):
|
||||
# Missing/malformed records are evidence failure, not a default pass.
|
||||
return False
|
||||
|
||||
|
||||
def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Signals:
|
||||
if not _has_complete_evidence(baseline) or not _has_complete_evidence(candidate):
|
||||
return Signals(
|
||||
surface_differs=False, realization_failed=False, postcondition_met=None,
|
||||
verdicts_regressed=False, verdicts_inconclusive=False, claims_differ=False,
|
||||
evidence_incomplete=True,
|
||||
)
|
||||
before, after = _verdict_map(baseline), _verdict_map(candidate)
|
||||
|
||||
regressed = any(
|
||||
|
|
@ -167,7 +224,10 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -
|
|||
verdicts_regressed=regressed,
|
||||
verdicts_inconclusive=any(v == "INCONCLUSIVE" for v in after.values()),
|
||||
claims_differ=_claim_fingerprint(baseline) != _claim_fingerprint(candidate),
|
||||
evidence_incomplete=not after or not candidate.get("sut_version"),
|
||||
evidence_incomplete=(
|
||||
not before.keys() <= after.keys()
|
||||
or not set(baseline["scheduled_steps"]) <= set(candidate["scheduled_steps"])
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -192,13 +252,13 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco
|
|||
change that happens to break a claim.
|
||||
"""
|
||||
signals = extract_signals(baseline, candidate)
|
||||
failures = regressions(baseline, candidate)
|
||||
|
||||
# 1. Evidence first. A conclusion drawn from incomplete evidence is worse
|
||||
# than no conclusion, in either direction.
|
||||
if signals.evidence_incomplete:
|
||||
return Outcome(Classification.AMBIGUOUS,
|
||||
"the run did not retain enough evidence to classify", signals)
|
||||
failures = regressions(baseline, candidate)
|
||||
if signals.verdicts_inconclusive:
|
||||
return Outcome(Classification.AMBIGUOUS,
|
||||
"at least one assertion could not be judged from the evidence",
|
||||
|
|
|
|||
|
|
@ -64,6 +64,9 @@ class EvidencePack:
|
|||
observations: list[Observation] = field(default_factory=list)
|
||||
verdicts: list[dict[str, Any]] = field(default_factory=list)
|
||||
provenance_index: dict[str, str] = field(default_factory=dict)
|
||||
# Run inputs, captured before execution. A surviving prefix is not a manifest.
|
||||
scheduled_steps: list[str] = field(default_factory=list)
|
||||
expected_judgments: list[dict[str, str]] = field(default_factory=list)
|
||||
|
||||
def record(self, observation: Observation) -> None:
|
||||
self.observations.append(observation)
|
||||
|
|
|
|||
|
|
@ -118,6 +118,15 @@ class Runner:
|
|||
scenario_id=scenario.id,
|
||||
use_case_id=scenario.use_case.id,
|
||||
sut_version=self._world.sut_version,
|
||||
scheduled_steps=[step.id for step in scenario.steps],
|
||||
expected_judgments=[
|
||||
{"assertion_id": assertion.id, "step_id": step.id}
|
||||
for step in scenario.steps
|
||||
for assertion in (
|
||||
*scenario.use_case.invariants,
|
||||
*(c for c in scenario.use_case.claims if c.after_step == step.id),
|
||||
)
|
||||
],
|
||||
)
|
||||
pack.provenance_index = {
|
||||
scenario.use_case.id: scenario.use_case.provenance.value,
|
||||
|
|
@ -133,11 +142,12 @@ class Runner:
|
|||
|
||||
judgments: list[Judgment] = []
|
||||
scenario_sound = True
|
||||
aborted = False
|
||||
claims_by_step: dict[str, list] = {}
|
||||
for claim in scenario.use_case.claims:
|
||||
claims_by_step.setdefault(claim.after_step, []).append(claim)
|
||||
|
||||
for step in scenario.steps:
|
||||
for step_index, step in enumerate(scenario.steps):
|
||||
actor = self._world.cast[step.actor_id]
|
||||
|
||||
# --- S1: how it was done -------------------------------------
|
||||
|
|
@ -151,6 +161,21 @@ class Runner:
|
|||
{"step": step.id, "action": step.action.describe(), "error": str(exc)},
|
||||
step.id,
|
||||
)
|
||||
aborted = True
|
||||
# Keep the scheduled assertion set, but never invent observations
|
||||
# for steps we did not execute. Claims outside this scenario's
|
||||
# schedule remain outside its scope (e.g. the one-step browser arm).
|
||||
for skipped in scenario.steps[step_index:]:
|
||||
for assertion in (
|
||||
*scenario.use_case.invariants,
|
||||
*claims_by_step.get(skipped.id, ()),
|
||||
):
|
||||
judgments.append(Judgment(
|
||||
assertion.id, assertion.text, Verdict.INCONCLUSIVE,
|
||||
skipped.id,
|
||||
{"reason": f"run aborted at {step.id!r}; "
|
||||
"this scheduled step was not executed"},
|
||||
))
|
||||
break
|
||||
|
||||
self._record(
|
||||
|
|
@ -225,4 +250,8 @@ class Runner:
|
|||
pack.finished_at = datetime.now(timezone.utc).isoformat()
|
||||
|
||||
result_verdict = overall(judgments)
|
||||
# Even an assertion-free trailing step is part of the scheduled run.
|
||||
# Preserve observed failures, but a passing prefix cannot certify an abort.
|
||||
if aborted and result_verdict is Verdict.PASS:
|
||||
result_verdict = Verdict.INCONCLUSIVE
|
||||
return RunResult(run_id, result_verdict, judgments, pack)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue