Require qualified runs and intent revisions for automatic acceptance
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
parent
7779768058
commit
a8bb787d12
10 changed files with 486 additions and 14 deletions
228
tests/test_acceptance_boundaries.py
Normal file
228
tests/test_acceptance_boundaries.py
Normal file
|
|
@ -0,0 +1,228 @@
|
|||
"""Admission, crystallization and intent revisions must agree on safe evidence."""
|
||||
from copy import deepcopy
|
||||
from dataclasses import replace
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from scenarios.alice_bob_carol import build
|
||||
from testdriver import Runner
|
||||
from testdriver.classification import Classification, classify
|
||||
from testdriver.crystallization import assess_stability
|
||||
|
||||
|
||||
def pack(*mutations, change=None, abort=False):
|
||||
world, driver, observer, asset, oracle = build(*mutations)
|
||||
if change:
|
||||
asset.scenario = replace(asset.scenario, use_case=change(asset.scenario.use_case))
|
||||
if abort:
|
||||
steps = list(asset.scenario.steps)
|
||||
steps[1] = replace(steps[1], action=replace(
|
||||
steps[1].action, permitted_surfaces=frozenset({"browser"})))
|
||||
asset.scenario = replace(asset.scenario, steps=tuple(steps))
|
||||
return json.loads(Runner(world, driver, observer, oracle).run(asset).evidence.to_json())
|
||||
|
||||
|
||||
def change_claim(**changes):
|
||||
return lambda case: replace(case, claims=tuple(
|
||||
replace(c, **changes) if c.id == "c-bob-revoked" else c for c in case.claims
|
||||
))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("mutation", ["M17", "M15", "M16", "M18"])
|
||||
def test_failing_baseline_cannot_authorize_same_failure_or_fixed_run(mutation):
|
||||
failing = pack(mutation)
|
||||
assert any(v["verdict"] == "FAIL" for v in failing["verdicts"])
|
||||
for candidate in (deepcopy(failing), pack()):
|
||||
outcome = classify(failing, candidate)
|
||||
assert not outcome.safe_to_accept
|
||||
assert outcome.classification is Classification.AMBIGUOUS
|
||||
assert "baseline" in outcome.reason
|
||||
|
||||
|
||||
@pytest.mark.parametrize("damage", ["abort", "snapshots", "verdict", "failure", "inconclusive"])
|
||||
def test_repeated_invalid_runs_cannot_crystallize(damage):
|
||||
packs = [pack("M17") if damage == "failure" else pack(abort=damage == "abort")
|
||||
for _ in range(3)]
|
||||
for candidate in packs:
|
||||
if damage == "snapshots":
|
||||
candidate["observations"] = [o for o in candidate["observations"]
|
||||
if o["kind"] != "state_snapshot"]
|
||||
elif damage == "verdict":
|
||||
candidate["verdicts"].pop()
|
||||
elif damage == "inconclusive":
|
||||
candidate["verdicts"][0]["verdict"] = "INCONCLUSIVE"
|
||||
report = assess_stability(packs)
|
||||
assert not report.stable
|
||||
assert not report.trajectories
|
||||
assert report.reason
|
||||
|
||||
|
||||
def test_one_bad_run_in_an_otherwise_stable_window_prevents_freezing():
|
||||
packs = [pack(), pack(abort=True), pack()]
|
||||
assert not assess_stability(packs).stable
|
||||
|
||||
|
||||
def test_replaying_one_receipt_does_not_count_as_three_runs():
|
||||
baseline = pack()
|
||||
assert not assess_stability([deepcopy(baseline) for _ in range(3)]).stable
|
||||
|
||||
|
||||
def test_complete_passing_runs_still_classify_and_crystallize():
|
||||
packs = [pack() for _ in range(3)]
|
||||
assert classify(packs[0], packs[1]).classification is Classification.UNCHANGED
|
||||
assert assess_stability(packs).stable
|
||||
|
||||
|
||||
@pytest.mark.parametrize("changes", [
|
||||
{"text": "Revocation is optional"},
|
||||
{"source_ref": "new-independent-spec/revocation-v2"},
|
||||
{"predicate": lambda obs: True},
|
||||
])
|
||||
def test_same_id_does_not_hide_a_changed_claim(changes):
|
||||
before, after = pack(), pack(change=change_claim(**changes))
|
||||
assert classify(before, after).classification is Classification.INTENT_CHANGED
|
||||
assert not assess_stability([before, after, pack()]).stable
|
||||
|
||||
|
||||
def test_invariant_changes_are_intent_changes_too():
|
||||
def change(case):
|
||||
return replace(case, invariants=(replace(case.invariants[0], predicate=lambda obs: True),
|
||||
*case.invariants[1:]))
|
||||
assert classify(pack(), pack(change=change)).classification is Classification.INTENT_CHANGED
|
||||
|
||||
|
||||
def test_narrative_changes_require_intent_review():
|
||||
assert classify(pack(), pack(change=lambda c: replace(c, narrative="New policy"))) \
|
||||
.classification is Classification.INTENT_CHANGED
|
||||
|
||||
|
||||
def threshold(value):
|
||||
return lambda obs: len(obs["audit:R"]) >= value
|
||||
|
||||
|
||||
def default_threshold(value):
|
||||
return lambda obs, floor=value: len(obs["audit:R"]) >= floor
|
||||
|
||||
|
||||
@pytest.mark.parametrize("factory", [threshold, default_threshold])
|
||||
def test_captured_predicate_values_are_part_of_the_revision(factory):
|
||||
first = pack(change=change_claim(predicate=factory(0)))
|
||||
second = pack(change=change_claim(predicate=factory(1)))
|
||||
assert classify(first, second).classification is Classification.INTENT_CHANGED
|
||||
assert classify(first, pack(change=change_claim(predicate=factory(0)))).safe_to_accept
|
||||
|
||||
|
||||
POLICY_FLOOR = 0
|
||||
|
||||
|
||||
def helper(obs):
|
||||
return len(obs["audit:R"]) >= POLICY_FLOOR
|
||||
|
||||
|
||||
def predicate_using_helper(obs):
|
||||
return helper(obs)
|
||||
|
||||
|
||||
def test_global_values_and_helpers_are_part_of_the_revision(monkeypatch):
|
||||
first = pack(change=change_claim(predicate=predicate_using_helper))
|
||||
monkeypatch.setitem(globals(), "POLICY_FLOOR", 1)
|
||||
second = pack(change=change_claim(predicate=predicate_using_helper))
|
||||
assert classify(first, second).classification is Classification.INTENT_CHANGED
|
||||
monkeypatch.setitem(globals(), "helper", lambda obs: True)
|
||||
third = pack(change=change_claim(predicate=predicate_using_helper))
|
||||
assert classify(second, third).classification is Classification.INTENT_CHANGED
|
||||
|
||||
|
||||
def test_unidentifiable_callable_cannot_be_automatically_accepted():
|
||||
class Opaque:
|
||||
def __call__(self, obs):
|
||||
return True
|
||||
packs = [pack(change=change_claim(predicate=Opaque())) for _ in range(3)]
|
||||
assert not classify(packs[0], packs[1]).safe_to_accept
|
||||
assert not assess_stability(packs).stable
|
||||
|
||||
|
||||
def test_legacy_pack_without_intent_revisions_requires_rerun():
|
||||
before, after = pack(), pack()
|
||||
after.pop("intent_revisions", None)
|
||||
assert not classify(before, after).safe_to_accept
|
||||
assert not assess_stability([after, pack(), pack()]).stable
|
||||
|
||||
|
||||
def test_predicate_location_does_not_change_revision():
|
||||
from types import FunctionType
|
||||
from scenarios.alice_bob_carol import USE_CASE
|
||||
from testdriver.revisions import intent_revisions
|
||||
|
||||
original = USE_CASE.claims[0].predicate
|
||||
moved = FunctionType(original.__code__.replace(
|
||||
co_filename="/a/different/checkout/scenario.py", co_firstlineno=999,
|
||||
), original.__globals__)
|
||||
changed = replace(USE_CASE, claims=(replace(USE_CASE.claims[0], predicate=moved),
|
||||
*USE_CASE.claims[1:]))
|
||||
assert intent_revisions(USE_CASE) == intent_revisions(changed)
|
||||
|
||||
|
||||
def test_revisions_are_stable_across_processes():
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from scenarios.alice_bob_carol import USE_CASE
|
||||
from testdriver.revisions import intent_revisions
|
||||
|
||||
root = Path(__file__).resolve().parents[1]
|
||||
output = subprocess.check_output([
|
||||
sys.executable, "-c",
|
||||
"import json; from scenarios.alice_bob_carol import USE_CASE; "
|
||||
"from testdriver.revisions import intent_revisions; "
|
||||
"print(json.dumps(intent_revisions(USE_CASE)))",
|
||||
], cwd=root, env={**os.environ, "PYTHONPATH": os.pathsep.join((str(root / "src"), str(root)))},
|
||||
text=True)
|
||||
assert json.loads(output) == intent_revisions(USE_CASE)
|
||||
|
||||
|
||||
def test_captured_values_are_hashed_not_retained():
|
||||
secret = "synthetic-private-dependency-do-not-retain"
|
||||
predicate = lambda obs: bool(secret) and obs["probe_read:bob:R"] is False
|
||||
result = pack(change=change_claim(predicate=predicate))
|
||||
assert result["intent_revisions"]["c-bob-revoked"] is not None
|
||||
assert secret not in json.dumps(result)
|
||||
|
||||
|
||||
def test_schedule_change_is_an_intent_change():
|
||||
before = pack()
|
||||
after = pack(change=change_claim(after_step="s1-create", predicate=lambda obs: True))
|
||||
assert classify(before, after).classification is Classification.INTENT_CHANGED
|
||||
|
||||
|
||||
def test_failed_baseline_realization_does_not_authorize_automatic_acceptance():
|
||||
before, after = pack(), pack()
|
||||
next(o for o in before["observations"]
|
||||
if o["kind"] == "realization")["data"]["raised"] = "Unrealized"
|
||||
assert not classify(before, after).safe_to_accept
|
||||
|
||||
|
||||
def test_unknown_baseline_postcondition_does_not_authorize_automatic_acceptance():
|
||||
before, after = pack(), pack()
|
||||
next(o for o in before["observations"]
|
||||
if o["kind"] == "realization_check")["data"]["postcondition_met"] = None
|
||||
assert not classify(before, after).safe_to_accept
|
||||
|
||||
|
||||
def test_dynamic_global_lookup_is_not_given_a_misleading_revision():
|
||||
def dynamic(obs):
|
||||
return globals()["POLICY_FLOOR"] <= len(obs["audit:R"])
|
||||
before, after = pack(change=change_claim(predicate=dynamic)), pack(change=change_claim(predicate=dynamic))
|
||||
assert before["intent_revisions"]["c-bob-revoked"] is None
|
||||
assert not classify(before, after).safe_to_accept
|
||||
|
||||
|
||||
@pytest.mark.parametrize("invalid", [None, "true", 1])
|
||||
def test_one_unverified_postcondition_prevents_acceptance_and_freezing(invalid):
|
||||
runs = [pack() for _ in range(3)]
|
||||
next(o for o in runs[1]["observations"]
|
||||
if o["kind"] == "realization_check")["data"]["postcondition_met"] = invalid
|
||||
assert not classify(runs[0], runs[1]).safe_to_accept
|
||||
assert not assess_stability(runs).stable
|
||||
Loading…
Add table
Add a link
Reference in a new issue