All four gate criteria met. False Adaptation Rate 0/7 with 12 of 13 mechanical mutations absorbed. 178 tests pass. TD-WP-0002 finished. Fitness loop closed via F-0003: actor isolation was a property of scenarios written to expose it, not of runs. Actors now carry an automatic private marker and the runner examines all of them on every scenario, with two permanent regressions behind it. Compression - six abstractions removed, each declared and never used: Verdict.SUSPICIOUS (a verdict no oracle could emit), Step.expect_refusal, ActorIsolationError, World.seed, EvidencePack.latest, Trajectory.method. F-0008: Temperature may be redundant. Crystallization was built without it ever being consulted; measured stability of realization did the work, and is observed rather than declared. Gated for removal alongside energy.py. INTENT_CHANGED and REALIZATION_FAILED had never run. Both now have purpose-built cases and a test that fails if a seventh outcome is added without one. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Assistant: claude-code Assistant-Model: opus Assistant-Process: 1629012@bnt-lap001 Assistant-Session: 78d4fb13-8a1e-474b-87a3-9b9261c49a39
119 lines
4 KiB
Python
119 lines
4 KiB
Python
"""The framework's foundational guarantees, verified from outside the framework.
|
|
|
|
These read an Evidence Pack the way an auditor would — as JSON, with no live
|
|
objects — and use plain pytest. No Oracle, no Runner, no Verdict aggregation.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
|
|
import pytest
|
|
|
|
from testdriver import Runner
|
|
from scenarios.alice_bob_carol import USE_CASE, build
|
|
from tests.selfverification.checks import (
|
|
SELF_CHECKS,
|
|
check_actor_isolation,
|
|
check_evidence_supports_every_verdict,
|
|
check_intent_independence,
|
|
check_isolation_was_examined,
|
|
check_no_actor_collected_judgment,
|
|
check_runs_agree,
|
|
check_verdicts_follow_from_judgment_evidence,
|
|
)
|
|
|
|
SECRETS = {
|
|
"alice": "alice-only-9f3c",
|
|
"bob": "bob-only-2a71",
|
|
"carol": "carol-only-55e0",
|
|
}
|
|
|
|
|
|
def run_and_serialize(*mutations: str):
|
|
"""Run, then throw away everything except the serialized evidence.
|
|
|
|
Deliberate: the checks must work from the artefact alone, exactly as they
|
|
would months later, with no access to the objects that produced it.
|
|
"""
|
|
world, driver, observer, asset, oracle = build(*mutations)
|
|
for actor_id, secret in SECRETS.items():
|
|
world.cast[actor_id].remember("private", secret)
|
|
result = Runner(world, driver, observer, oracle).run(asset)
|
|
pack = json.loads(result.evidence.to_json())
|
|
memories = {a.id: {k: a.recall(k) for k in a.known_keys()} for a in world.cast}
|
|
return pack, memories, list(world.cast.actors)
|
|
|
|
|
|
def test_the_registry_is_complete():
|
|
assert len(SELF_CHECKS) == 4
|
|
|
|
|
|
# --- td://self/actor-isolation -------------------------------------------
|
|
|
|
|
|
def test_actor_isolation():
|
|
pack, memories, _ = run_and_serialize()
|
|
assert check_actor_isolation(pack, SECRETS, memories) == []
|
|
|
|
|
|
@pytest.mark.parametrize("mutations", [(), ("M15",), ("M01",)])
|
|
def test_every_run_examines_isolation(mutations):
|
|
"""F-0003 resolved: isolation is now a property of runs, not of scenarios
|
|
written to expose it. No canaries need be planted by the test."""
|
|
pack, _, _ = run_and_serialize(*mutations)
|
|
assert check_isolation_was_examined(pack) == []
|
|
|
|
|
|
# --- td://self/oracle-independence ---------------------------------------
|
|
|
|
|
|
def test_no_actor_collected_judgment_evidence():
|
|
pack, _, actors = run_and_serialize()
|
|
assert check_no_actor_collected_judgment(pack, actors) == []
|
|
|
|
|
|
@pytest.mark.parametrize("mutations", [(), ("M15",), ("M11",), ("M01", "M15")])
|
|
def test_verdicts_are_reproducible_from_judgment_evidence_alone(mutations):
|
|
"""Holds on passing and failing runs alike — a check that only works when
|
|
everything is green is not verifying independence, it is verifying luck."""
|
|
pack, _, _ = run_and_serialize(*mutations)
|
|
assertions = list(USE_CASE.claims) + list(USE_CASE.invariants)
|
|
assert check_verdicts_follow_from_judgment_evidence(pack, assertions) == []
|
|
|
|
|
|
# --- td://self/evidence-reproducibility -----------------------------------
|
|
|
|
|
|
def test_every_verdict_is_supported_by_retained_evidence():
|
|
pack, _, _ = run_and_serialize()
|
|
assert check_evidence_supports_every_verdict(pack) == []
|
|
|
|
|
|
def test_two_runs_from_the_same_seed_agree():
|
|
first, _, _ = run_and_serialize()
|
|
second, _, _ = run_and_serialize()
|
|
assert check_runs_agree(first, second) == []
|
|
assert first["run_id"] != second["run_id"]
|
|
|
|
|
|
def test_reproducibility_holds_on_a_failing_run():
|
|
first, _, _ = run_and_serialize("M15")
|
|
second, _, _ = run_and_serialize("M15")
|
|
assert check_runs_agree(first, second) == []
|
|
assert any(v["verdict"] == "FAIL" for v in first["verdicts"])
|
|
|
|
|
|
# --- td://self/intent-independence ----------------------------------------
|
|
|
|
|
|
def test_intent_independence():
|
|
pack, _, _ = run_and_serialize()
|
|
assert check_intent_independence(pack) == []
|
|
|
|
|
|
def test_intent_independence_holds_when_a_claim_fails():
|
|
"""The moment that matters: a FAIL is an accusation, and an accusation
|
|
resting on implementation-derived intent is worthless."""
|
|
pack, _, _ = run_and_serialize("M15")
|
|
assert check_intent_independence(pack) == []
|