Require qualified runs and intent revisions for automatic acceptance

Assistant: codex
Assistant-Model: gpt-6-astra
Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
tegwick 2026-09-28 12:30:13 +02:00
parent 7779768058
commit a8bb787d12
10 changed files with 486 additions and 14 deletions

View file

@ -17,9 +17,9 @@ M19 in the lab prove it), so no amount of evidence separates them. What the
classifier can honestly say is *"behaviour changed against intent that did not"*,
and hand that to a human.
Intent change **is** detectable, but only when a human has actually changed the
intent: the claim fingerprint moves. That is a fact about the recorded use case,
not an inference about behaviour.
Intent change **is** detectable when recorded definitions change: the intent
revision moves. That calls for review; it does not establish who approved the
change and is not an inference about behaviour.
"""
from __future__ import annotations
@ -45,7 +45,7 @@ class Classification(str, Enum):
"""
INTENT_CHANGED = "INTENT_CHANGED"
"""The recorded claim set itself changed. A human has already acted."""
"""Recorded intent changed. Review its definitions and approval provenance."""
REALIZATION_FAILED = "REALIZATION_FAILED"
"""The action could not be performed at all. Not a verdict about the system."""
@ -132,7 +132,8 @@ def _surface_fingerprint(pack: Mapping[str, Any]) -> str:
def _claim_fingerprint(pack: Mapping[str, Any]) -> str:
return json.dumps(sorted(pack.get("provenance_index", {}).items()))
return json.dumps({"provenance": pack["provenance_index"],
"revisions": pack["intent_revisions"]}, sort_keys=True)
def _has_complete_evidence(pack: Mapping[str, Any]) -> bool:
@ -180,7 +181,15 @@ def _has_complete_evidence(pack: Mapping[str, Any]) -> bool:
return False
if any(o["kind"] == "surface_violation" for o in observations):
return False
return isinstance(pack["provenance_index"], Mapping)
revisions = pack["intent_revisions"]
required_ids = {assertion for assertion, _ in expected_keys} | {pack["use_case_id"]}
return (
isinstance(pack["provenance_index"], Mapping)
and isinstance(revisions, Mapping) and required_ids <= revisions.keys()
and all(isinstance(revision, str) and len(revision) == 64
and all(c in "0123456789abcdef" for c in revision)
for revision in revisions.values())
)
except (KeyError, TypeError, ValueError):
# Missing/malformed records are evidence failure, not a default pass.
return False
@ -194,6 +203,7 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -
evidence_incomplete=True,
)
before, after = _verdict_map(baseline), _verdict_map(candidate)
claims_differ = _claim_fingerprint(baseline) != _claim_fingerprint(candidate)
regressed = any(
after.get(key) == "FAIL" and verdict != "FAIL" for key, verdict in before.items()
@ -212,10 +222,10 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -
met = None
elif any(c.get("postcondition_met") is False for c in checks):
met = False
elif any(c.get("postcondition_met") is None for c in checks):
met = None
else:
elif all(c.get("postcondition_met") is True for c in checks):
met = True
else:
met = None
return Signals(
surface_differs=_surface_fingerprint(baseline) != _surface_fingerprint(candidate),
@ -223,10 +233,12 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -
postcondition_met=met,
verdicts_regressed=regressed,
verdicts_inconclusive=any(v == "INCONCLUSIVE" for v in after.values()),
claims_differ=_claim_fingerprint(baseline) != _claim_fingerprint(candidate),
claims_differ=claims_differ,
evidence_incomplete=(
not before.keys() <= after.keys()
or not set(baseline["scheduled_steps"]) <= set(candidate["scheduled_steps"])
not claims_differ and (
not before.keys() <= after.keys()
or not set(baseline["scheduled_steps"]) <= set(candidate["scheduled_steps"])
)
),
)
@ -258,6 +270,10 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco
if signals.evidence_incomplete:
return Outcome(Classification.AMBIGUOUS,
"the run did not retain enough evidence to classify", signals)
if any(v["verdict"] != "PASS" for v in baseline["verdicts"]):
return Outcome(Classification.AMBIGUOUS,
"the baseline is not a passing reference; obtain an accepted baseline",
signals)
failures = regressions(baseline, candidate)
if signals.verdicts_inconclusive:
return Outcome(Classification.AMBIGUOUS,
@ -267,8 +283,8 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco
# 2. Intent moving is a fact about the recorded use case, not an inference.
if signals.claims_differ:
return Outcome(Classification.INTENT_CHANGED,
"the recorded claim set differs from the baseline; "
"a human changed what is being asserted", signals, failures)
"the recorded intent differs from the baseline; "
"review the changed definitions and provenance", signals, failures)
# 3. A regression is reported before anything is allowed to explain it away.
# This is decision-table row 3: coincidence is not exoneration.
@ -295,6 +311,15 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco
return Outcome(Classification.AMBIGUOUS,
"the action's postcondition could not be evaluated", signals)
# A passing verdict alone does not prove the reference actions took effect.
baseline_checks = [o["data"] for o in baseline["observations"]
if o["kind"] == "realization_check"]
if (any(c.get("postcondition_met") is not True for c in baseline_checks)
or any(o["kind"] == "realization" and o["data"].get("raised")
for o in baseline["observations"])):
return Outcome(Classification.AMBIGUOUS,
"the baseline does not establish successful realization", signals)
# 6. Only now, with every claim intact and the action verified, may a
# surface difference be called an adaptation.
if signals.surface_differs:

View file

@ -24,6 +24,7 @@ from datetime import datetime, timezone
from typing import Any, Mapping, Sequence
from .actions import SemanticAction
from .classification import classify
from .drivers import Realization
from .world import Actor
@ -90,6 +91,20 @@ def assess_stability(packs: Sequence[Mapping[str, Any]], minimum: int = 3) -> St
False, len(packs), 0,
f"need at least {minimum} runs to judge stability, have {len(packs)}",
)
if not packs:
return StabilityReport(False, 0, 0, "no runs to judge stability")
for pack in packs:
outcome = classify(packs[0], pack)
if not outcome.safe_to_accept:
return StabilityReport(False, len(packs), 0,
f"run is not eligible for crystallization: {outcome.reason}")
if (pack["scenario_id"], pack["use_case_id"], pack["scheduled_steps"],
pack["expected_judgments"]) != (
packs[0]["scenario_id"], packs[0]["use_case_id"], packs[0]["scheduled_steps"],
packs[0]["expected_judgments"]):
return StabilityReport(False, len(packs), 0, "scenario scope differs across runs")
if len({pack["run_id"] for pack in packs}) != len(packs):
return StabilityReport(False, len(packs), 0, "need distinct runs, not repeated receipts")
captured = [capture(pack) for pack in packs]
keys = {tuple(t.key() for t in trajectory) for trajectory in captured}
if len(keys) != 1:

View file

@ -67,6 +67,7 @@ class EvidencePack:
# Run inputs, captured before execution. A surviving prefix is not a manifest.
scheduled_steps: list[str] = field(default_factory=list)
expected_judgments: list[dict[str, str]] = field(default_factory=list)
intent_revisions: dict[str, str | None] = field(default_factory=dict)
def record(self, observation: Observation) -> None:
self.observations.append(observation)

113
src/testdriver/revisions.py Normal file
View file

@ -0,0 +1,113 @@
"""Conservative, process-independent revisions of Python intent inputs.
Only hashes leave this module: captured values may be sensitive. Unsupported
runtime dependencies produce no revision, never a name-only fallback. This is
change detection for pure predicates, not a proof of semantic equivalence.
"""
from __future__ import annotations
import builtins
import dis
import hashlib
import json
import sys
from types import CodeType, FunctionType
from .intent import UseCase
# These can access dependencies that bytecode name inspection cannot enumerate.
_DYNAMIC_BUILTINS = frozenset({
"eval", "exec", "globals", "locals", "vars", "getattr", "setattr", "delattr",
"__import__", "open", "input",
})
def _dump(value) -> str:
return json.dumps(value, sort_keys=True, separators=(",", ":"))
def _global_names(code: CodeType) -> set[str]:
names = {i.argval for i in dis.get_instructions(code)
if i.opname in ("LOAD_GLOBAL", "LOAD_NAME")}
for constant in code.co_consts:
if isinstance(constant, CodeType):
names.update(_global_names(constant))
return names
def _describe(value, active: set[int]):
"""Do not execute predicates, getters, reprs or arbitrary serialization hooks."""
if value is None or type(value) in (bool, int, str):
return [type(value).__name__, value]
if type(value) is float:
return ["float", value.hex()]
if type(value) is bytes:
return ["bytes", value.hex()]
if value is Ellipsis:
return ["ellipsis"]
if len(active) >= 32 or id(value) in active:
raise ValueError("recursive or overly deep predicate dependency")
active = active | {id(value)}
if type(value) in (tuple, list, set, frozenset):
items = [_describe(item, active) for item in value]
if type(value) in (set, frozenset):
items.sort(key=_dump)
return [type(value).__name__, items]
if type(value) is dict:
items = [[_describe(k, active), _describe(v, active)] for k, v in value.items()]
return ["dict", sorted(items, key=_dump)]
if isinstance(value, CodeType):
return ["code", value.co_code.hex(), _describe(value.co_consts, active),
value.co_names, value.co_varnames, value.co_freevars, value.co_cellvars,
value.co_argcount, value.co_posonlyargcount, value.co_kwonlyargcount,
value.co_flags, value.co_exceptiontable.hex()]
if isinstance(value, FunctionType):
dependencies = {}
for name in sorted(_global_names(value.__code__)):
if name in value.__globals__:
dependency = value.__globals__[name]
else:
dependency = value.__builtins__[name]
# Builtins are interpreter-version-bound, not opaque user callables.
if name in vars(builtins) and dependency is vars(builtins)[name]:
if name in _DYNAMIC_BUILTINS:
raise ValueError("dynamic predicate dependency")
dependencies[name] = ["builtin", name]
else:
dependencies[name] = _describe(dependency, active)
return ["function", _describe(value.__code__, active),
_describe(value.__defaults__, active), _describe(value.__kwdefaults__, active),
[_describe(cell.cell_contents, active) for cell in value.__closure__ or ()],
dependencies]
raise ValueError("unsupported predicate dependency")
def _digest(value) -> str:
return hashlib.sha256(_dump(value).encode()).hexdigest()
def intent_revisions(case: UseCase) -> dict[str, str | None]:
"""Capture intent before execution, including unscheduled use-case claims.
IDs remain stable; revisions change with their protected meaning. Predicate
bytecode is bound to the Python implementation/version. Code location and
line numbers are deliberately excluded so checkout paths do not cause drift.
"""
revisions: dict[str, str | None] = {
case.id: _digest(["use-case", case.title, case.narrative,
case.provenance.value, case.source_ref]),
}
for assertion in (*case.claims, *case.invariants):
if assertion.id in revisions:
return {} # Ambiguous identities cannot produce an admissible revision.
try:
predicate = _describe(assertion.predicate, set())
except (KeyError, TypeError, ValueError, RecursionError):
revisions[assertion.id] = None
continue
revisions[assertion.id] = _digest([
"python-intent-v1", sys.implementation.name, list(sys.version_info[:3]),
type(assertion).__name__, assertion.text, assertion.provenance.value,
assertion.source_ref, getattr(assertion, "after_step", None), predicate,
])
return revisions

View file

@ -18,6 +18,7 @@ from .drivers import Driver
from .evidence import EvidencePack, Observation, Stratum
from .observers import StateObserver
from .oracles import Judgment, Oracle, Verdict, overall
from .revisions import intent_revisions
from .scenario import Scenario, VerificationAsset
from .world import World
@ -118,6 +119,7 @@ class Runner:
scenario_id=scenario.id,
use_case_id=scenario.use_case.id,
sut_version=self._world.sut_version,
intent_revisions=intent_revisions(scenario.use_case),
scheduled_steps=[step.id for step in scenario.steps],
expected_judgments=[
{"assertion_id": assertion.id, "step_id": step.id}