Require qualified runs and intent revisions for automatic acceptance
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
parent
7779768058
commit
a8bb787d12
10 changed files with 486 additions and 14 deletions
|
|
@ -17,9 +17,9 @@ M19 in the lab prove it), so no amount of evidence separates them. What the
|
|||
classifier can honestly say is *"behaviour changed against intent that did not"*,
|
||||
and hand that to a human.
|
||||
|
||||
Intent change **is** detectable, but only when a human has actually changed the
|
||||
intent: the claim fingerprint moves. That is a fact about the recorded use case,
|
||||
not an inference about behaviour.
|
||||
Intent change **is** detectable when recorded definitions change: the intent
|
||||
revision moves. That calls for review; it does not establish who approved the
|
||||
change and is not an inference about behaviour.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -45,7 +45,7 @@ class Classification(str, Enum):
|
|||
"""
|
||||
|
||||
INTENT_CHANGED = "INTENT_CHANGED"
|
||||
"""The recorded claim set itself changed. A human has already acted."""
|
||||
"""Recorded intent changed. Review its definitions and approval provenance."""
|
||||
|
||||
REALIZATION_FAILED = "REALIZATION_FAILED"
|
||||
"""The action could not be performed at all. Not a verdict about the system."""
|
||||
|
|
@ -132,7 +132,8 @@ def _surface_fingerprint(pack: Mapping[str, Any]) -> str:
|
|||
|
||||
|
||||
def _claim_fingerprint(pack: Mapping[str, Any]) -> str:
|
||||
return json.dumps(sorted(pack.get("provenance_index", {}).items()))
|
||||
return json.dumps({"provenance": pack["provenance_index"],
|
||||
"revisions": pack["intent_revisions"]}, sort_keys=True)
|
||||
|
||||
|
||||
def _has_complete_evidence(pack: Mapping[str, Any]) -> bool:
|
||||
|
|
@ -180,7 +181,15 @@ def _has_complete_evidence(pack: Mapping[str, Any]) -> bool:
|
|||
return False
|
||||
if any(o["kind"] == "surface_violation" for o in observations):
|
||||
return False
|
||||
return isinstance(pack["provenance_index"], Mapping)
|
||||
revisions = pack["intent_revisions"]
|
||||
required_ids = {assertion for assertion, _ in expected_keys} | {pack["use_case_id"]}
|
||||
return (
|
||||
isinstance(pack["provenance_index"], Mapping)
|
||||
and isinstance(revisions, Mapping) and required_ids <= revisions.keys()
|
||||
and all(isinstance(revision, str) and len(revision) == 64
|
||||
and all(c in "0123456789abcdef" for c in revision)
|
||||
for revision in revisions.values())
|
||||
)
|
||||
except (KeyError, TypeError, ValueError):
|
||||
# Missing/malformed records are evidence failure, not a default pass.
|
||||
return False
|
||||
|
|
@ -194,6 +203,7 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -
|
|||
evidence_incomplete=True,
|
||||
)
|
||||
before, after = _verdict_map(baseline), _verdict_map(candidate)
|
||||
claims_differ = _claim_fingerprint(baseline) != _claim_fingerprint(candidate)
|
||||
|
||||
regressed = any(
|
||||
after.get(key) == "FAIL" and verdict != "FAIL" for key, verdict in before.items()
|
||||
|
|
@ -212,10 +222,10 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -
|
|||
met = None
|
||||
elif any(c.get("postcondition_met") is False for c in checks):
|
||||
met = False
|
||||
elif any(c.get("postcondition_met") is None for c in checks):
|
||||
met = None
|
||||
else:
|
||||
elif all(c.get("postcondition_met") is True for c in checks):
|
||||
met = True
|
||||
else:
|
||||
met = None
|
||||
|
||||
return Signals(
|
||||
surface_differs=_surface_fingerprint(baseline) != _surface_fingerprint(candidate),
|
||||
|
|
@ -223,10 +233,12 @@ def extract_signals(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -
|
|||
postcondition_met=met,
|
||||
verdicts_regressed=regressed,
|
||||
verdicts_inconclusive=any(v == "INCONCLUSIVE" for v in after.values()),
|
||||
claims_differ=_claim_fingerprint(baseline) != _claim_fingerprint(candidate),
|
||||
claims_differ=claims_differ,
|
||||
evidence_incomplete=(
|
||||
not before.keys() <= after.keys()
|
||||
or not set(baseline["scheduled_steps"]) <= set(candidate["scheduled_steps"])
|
||||
not claims_differ and (
|
||||
not before.keys() <= after.keys()
|
||||
or not set(baseline["scheduled_steps"]) <= set(candidate["scheduled_steps"])
|
||||
)
|
||||
),
|
||||
)
|
||||
|
||||
|
|
@ -258,6 +270,10 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco
|
|||
if signals.evidence_incomplete:
|
||||
return Outcome(Classification.AMBIGUOUS,
|
||||
"the run did not retain enough evidence to classify", signals)
|
||||
if any(v["verdict"] != "PASS" for v in baseline["verdicts"]):
|
||||
return Outcome(Classification.AMBIGUOUS,
|
||||
"the baseline is not a passing reference; obtain an accepted baseline",
|
||||
signals)
|
||||
failures = regressions(baseline, candidate)
|
||||
if signals.verdicts_inconclusive:
|
||||
return Outcome(Classification.AMBIGUOUS,
|
||||
|
|
@ -267,8 +283,8 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco
|
|||
# 2. Intent moving is a fact about the recorded use case, not an inference.
|
||||
if signals.claims_differ:
|
||||
return Outcome(Classification.INTENT_CHANGED,
|
||||
"the recorded claim set differs from the baseline; "
|
||||
"a human changed what is being asserted", signals, failures)
|
||||
"the recorded intent differs from the baseline; "
|
||||
"review the changed definitions and provenance", signals, failures)
|
||||
|
||||
# 3. A regression is reported before anything is allowed to explain it away.
|
||||
# This is decision-table row 3: coincidence is not exoneration.
|
||||
|
|
@ -295,6 +311,15 @@ def classify(baseline: Mapping[str, Any], candidate: Mapping[str, Any]) -> Outco
|
|||
return Outcome(Classification.AMBIGUOUS,
|
||||
"the action's postcondition could not be evaluated", signals)
|
||||
|
||||
# A passing verdict alone does not prove the reference actions took effect.
|
||||
baseline_checks = [o["data"] for o in baseline["observations"]
|
||||
if o["kind"] == "realization_check"]
|
||||
if (any(c.get("postcondition_met") is not True for c in baseline_checks)
|
||||
or any(o["kind"] == "realization" and o["data"].get("raised")
|
||||
for o in baseline["observations"])):
|
||||
return Outcome(Classification.AMBIGUOUS,
|
||||
"the baseline does not establish successful realization", signals)
|
||||
|
||||
# 6. Only now, with every claim intact and the action verified, may a
|
||||
# surface difference be called an adaptation.
|
||||
if signals.surface_differs:
|
||||
|
|
|
|||
|
|
@ -24,6 +24,7 @@ from datetime import datetime, timezone
|
|||
from typing import Any, Mapping, Sequence
|
||||
|
||||
from .actions import SemanticAction
|
||||
from .classification import classify
|
||||
from .drivers import Realization
|
||||
from .world import Actor
|
||||
|
||||
|
|
@ -90,6 +91,20 @@ def assess_stability(packs: Sequence[Mapping[str, Any]], minimum: int = 3) -> St
|
|||
False, len(packs), 0,
|
||||
f"need at least {minimum} runs to judge stability, have {len(packs)}",
|
||||
)
|
||||
if not packs:
|
||||
return StabilityReport(False, 0, 0, "no runs to judge stability")
|
||||
for pack in packs:
|
||||
outcome = classify(packs[0], pack)
|
||||
if not outcome.safe_to_accept:
|
||||
return StabilityReport(False, len(packs), 0,
|
||||
f"run is not eligible for crystallization: {outcome.reason}")
|
||||
if (pack["scenario_id"], pack["use_case_id"], pack["scheduled_steps"],
|
||||
pack["expected_judgments"]) != (
|
||||
packs[0]["scenario_id"], packs[0]["use_case_id"], packs[0]["scheduled_steps"],
|
||||
packs[0]["expected_judgments"]):
|
||||
return StabilityReport(False, len(packs), 0, "scenario scope differs across runs")
|
||||
if len({pack["run_id"] for pack in packs}) != len(packs):
|
||||
return StabilityReport(False, len(packs), 0, "need distinct runs, not repeated receipts")
|
||||
captured = [capture(pack) for pack in packs]
|
||||
keys = {tuple(t.key() for t in trajectory) for trajectory in captured}
|
||||
if len(keys) != 1:
|
||||
|
|
|
|||
|
|
@ -67,6 +67,7 @@ class EvidencePack:
|
|||
# Run inputs, captured before execution. A surviving prefix is not a manifest.
|
||||
scheduled_steps: list[str] = field(default_factory=list)
|
||||
expected_judgments: list[dict[str, str]] = field(default_factory=list)
|
||||
intent_revisions: dict[str, str | None] = field(default_factory=dict)
|
||||
|
||||
def record(self, observation: Observation) -> None:
|
||||
self.observations.append(observation)
|
||||
|
|
|
|||
113
src/testdriver/revisions.py
Normal file
113
src/testdriver/revisions.py
Normal file
|
|
@ -0,0 +1,113 @@
|
|||
"""Conservative, process-independent revisions of Python intent inputs.
|
||||
|
||||
Only hashes leave this module: captured values may be sensitive. Unsupported
|
||||
runtime dependencies produce no revision, never a name-only fallback. This is
|
||||
change detection for pure predicates, not a proof of semantic equivalence.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import builtins
|
||||
import dis
|
||||
import hashlib
|
||||
import json
|
||||
import sys
|
||||
from types import CodeType, FunctionType
|
||||
|
||||
from .intent import UseCase
|
||||
|
||||
# These can access dependencies that bytecode name inspection cannot enumerate.
|
||||
_DYNAMIC_BUILTINS = frozenset({
|
||||
"eval", "exec", "globals", "locals", "vars", "getattr", "setattr", "delattr",
|
||||
"__import__", "open", "input",
|
||||
})
|
||||
|
||||
|
||||
def _dump(value) -> str:
|
||||
return json.dumps(value, sort_keys=True, separators=(",", ":"))
|
||||
|
||||
|
||||
def _global_names(code: CodeType) -> set[str]:
|
||||
names = {i.argval for i in dis.get_instructions(code)
|
||||
if i.opname in ("LOAD_GLOBAL", "LOAD_NAME")}
|
||||
for constant in code.co_consts:
|
||||
if isinstance(constant, CodeType):
|
||||
names.update(_global_names(constant))
|
||||
return names
|
||||
|
||||
|
||||
def _describe(value, active: set[int]):
|
||||
"""Do not execute predicates, getters, reprs or arbitrary serialization hooks."""
|
||||
if value is None or type(value) in (bool, int, str):
|
||||
return [type(value).__name__, value]
|
||||
if type(value) is float:
|
||||
return ["float", value.hex()]
|
||||
if type(value) is bytes:
|
||||
return ["bytes", value.hex()]
|
||||
if value is Ellipsis:
|
||||
return ["ellipsis"]
|
||||
if len(active) >= 32 or id(value) in active:
|
||||
raise ValueError("recursive or overly deep predicate dependency")
|
||||
active = active | {id(value)}
|
||||
if type(value) in (tuple, list, set, frozenset):
|
||||
items = [_describe(item, active) for item in value]
|
||||
if type(value) in (set, frozenset):
|
||||
items.sort(key=_dump)
|
||||
return [type(value).__name__, items]
|
||||
if type(value) is dict:
|
||||
items = [[_describe(k, active), _describe(v, active)] for k, v in value.items()]
|
||||
return ["dict", sorted(items, key=_dump)]
|
||||
if isinstance(value, CodeType):
|
||||
return ["code", value.co_code.hex(), _describe(value.co_consts, active),
|
||||
value.co_names, value.co_varnames, value.co_freevars, value.co_cellvars,
|
||||
value.co_argcount, value.co_posonlyargcount, value.co_kwonlyargcount,
|
||||
value.co_flags, value.co_exceptiontable.hex()]
|
||||
if isinstance(value, FunctionType):
|
||||
dependencies = {}
|
||||
for name in sorted(_global_names(value.__code__)):
|
||||
if name in value.__globals__:
|
||||
dependency = value.__globals__[name]
|
||||
else:
|
||||
dependency = value.__builtins__[name]
|
||||
# Builtins are interpreter-version-bound, not opaque user callables.
|
||||
if name in vars(builtins) and dependency is vars(builtins)[name]:
|
||||
if name in _DYNAMIC_BUILTINS:
|
||||
raise ValueError("dynamic predicate dependency")
|
||||
dependencies[name] = ["builtin", name]
|
||||
else:
|
||||
dependencies[name] = _describe(dependency, active)
|
||||
return ["function", _describe(value.__code__, active),
|
||||
_describe(value.__defaults__, active), _describe(value.__kwdefaults__, active),
|
||||
[_describe(cell.cell_contents, active) for cell in value.__closure__ or ()],
|
||||
dependencies]
|
||||
raise ValueError("unsupported predicate dependency")
|
||||
|
||||
|
||||
def _digest(value) -> str:
|
||||
return hashlib.sha256(_dump(value).encode()).hexdigest()
|
||||
|
||||
|
||||
def intent_revisions(case: UseCase) -> dict[str, str | None]:
|
||||
"""Capture intent before execution, including unscheduled use-case claims.
|
||||
|
||||
IDs remain stable; revisions change with their protected meaning. Predicate
|
||||
bytecode is bound to the Python implementation/version. Code location and
|
||||
line numbers are deliberately excluded so checkout paths do not cause drift.
|
||||
"""
|
||||
revisions: dict[str, str | None] = {
|
||||
case.id: _digest(["use-case", case.title, case.narrative,
|
||||
case.provenance.value, case.source_ref]),
|
||||
}
|
||||
for assertion in (*case.claims, *case.invariants):
|
||||
if assertion.id in revisions:
|
||||
return {} # Ambiguous identities cannot produce an admissible revision.
|
||||
try:
|
||||
predicate = _describe(assertion.predicate, set())
|
||||
except (KeyError, TypeError, ValueError, RecursionError):
|
||||
revisions[assertion.id] = None
|
||||
continue
|
||||
revisions[assertion.id] = _digest([
|
||||
"python-intent-v1", sys.implementation.name, list(sys.version_info[:3]),
|
||||
type(assertion).__name__, assertion.text, assertion.provenance.value,
|
||||
assertion.source_ref, getattr(assertion, "after_step", None), predicate,
|
||||
])
|
||||
return revisions
|
||||
|
|
@ -18,6 +18,7 @@ from .drivers import Driver
|
|||
from .evidence import EvidencePack, Observation, Stratum
|
||||
from .observers import StateObserver
|
||||
from .oracles import Judgment, Oracle, Verdict, overall
|
||||
from .revisions import intent_revisions
|
||||
from .scenario import Scenario, VerificationAsset
|
||||
from .world import World
|
||||
|
||||
|
|
@ -118,6 +119,7 @@ class Runner:
|
|||
scenario_id=scenario.id,
|
||||
use_case_id=scenario.use_case.id,
|
||||
sut_version=self._world.sut_version,
|
||||
intent_revisions=intent_revisions(scenario.use_case),
|
||||
scheduled_steps=[step.id for step in scenario.steps],
|
||||
expected_judgments=[
|
||||
{"assertion_id": assertion.id, "step_id": step.id}
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue