T10: gate review and first compression pass
All four gate criteria met. False Adaptation Rate 0/7 with 12 of 13 mechanical mutations absorbed. 178 tests pass. TD-WP-0002 finished. Fitness loop closed via F-0003: actor isolation was a property of scenarios written to expose it, not of runs. Actors now carry an automatic private marker and the runner examines all of them on every scenario, with two permanent regressions behind it. Compression - six abstractions removed, each declared and never used: Verdict.SUSPICIOUS (a verdict no oracle could emit), Step.expect_refusal, ActorIsolationError, World.seed, EvidencePack.latest, Trajectory.method. F-0008: Temperature may be redundant. Crystallization was built without it ever being consulted; measured stability of realization did the work, and is observed rather than declared. Gated for removal alongside energy.py. INTENT_CHANGED and REALIZATION_FAILED had never run. Both now have purpose-built cases and a test that fails if a seventh outcome is added without one. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Assistant: claude-code Assistant-Model: opus Assistant-Process: 1629012@bnt-lap001 Assistant-Session: 78d4fb13-8a1e-474b-87a3-9b9261c49a39
This commit is contained in:
parent
4f4219d8f7
commit
1b9860a8ee
40 changed files with 1074 additions and 46 deletions
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
|
|
@ -94,6 +94,23 @@ def check_actor_isolation(
|
|||
return violations
|
||||
|
||||
|
||||
def check_isolation_was_examined(pack: Mapping[str, Any]) -> list[str]:
|
||||
"""Every run must carry a verdict on actor isolation — F-0003, resolved.
|
||||
|
||||
Before this existed, isolation was only observable in scenarios written to
|
||||
expose it: a run in which every actor shared one memory store produced
|
||||
evidence indistinguishable from a correct one. Actors now carry an automatic
|
||||
private marker and the runner examines them on every scenario, so the absence
|
||||
of this observation is itself a failure.
|
||||
"""
|
||||
for obs in _observations(pack):
|
||||
if obs["kind"] != "actor_isolation":
|
||||
continue
|
||||
violations = obs["data"].get("violations") or []
|
||||
return [f"actor isolation violated: {v}" for v in violations]
|
||||
return ["this run did not examine actor isolation at all"]
|
||||
|
||||
|
||||
# --- td://self/oracle-independence ---------------------------------------
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ from tests.selfverification.checks import (
|
|||
check_actor_isolation,
|
||||
check_evidence_supports_every_verdict,
|
||||
check_intent_independence,
|
||||
check_isolation_was_examined,
|
||||
check_no_actor_collected_judgment,
|
||||
check_runs_agree,
|
||||
check_verdicts_follow_from_judgment_evidence,
|
||||
|
|
@ -186,3 +187,35 @@ def test_unrecorded_provenance_is_caught():
|
|||
tampered = copy.deepcopy(pack)
|
||||
tampered["provenance_index"] = {}
|
||||
assert check_intent_independence(tampered)
|
||||
|
||||
|
||||
# --- F-0003 resolution: isolation observed on every run -------------------
|
||||
|
||||
|
||||
def test_an_unexamined_run_is_caught():
|
||||
"""A run that never looked at isolation must not read as isolated."""
|
||||
pack, _, _ = run_and_serialize()
|
||||
tampered = copy.deepcopy(pack)
|
||||
tampered["observations"] = [
|
||||
o for o in tampered["observations"] if o["kind"] != "actor_isolation"
|
||||
]
|
||||
assert check_isolation_was_examined(tampered)
|
||||
|
||||
|
||||
def test_a_leak_is_caught_without_the_test_planting_anything():
|
||||
"""The regression F-0003 leaves behind.
|
||||
|
||||
No secrets seeded by the harness, no scenario written to expose isolation.
|
||||
An actor holding another's automatic marker is caught by the ordinary run.
|
||||
"""
|
||||
from testdriver import Oracle, Runner
|
||||
from scenarios.alice_bob_carol import build
|
||||
|
||||
world, driver, observer, asset, oracle = build()
|
||||
world.cast["bob"].remember("overheard", world.cast["alice"].canary)
|
||||
result = Runner(world, driver, observer, oracle).run(asset)
|
||||
pack = json.loads(result.evidence.to_json())
|
||||
|
||||
violations = check_isolation_was_examined(pack)
|
||||
assert violations
|
||||
assert "holds the private marker of 'alice'" in violations[0]
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from tests.selfverification.checks import (
|
|||
check_actor_isolation,
|
||||
check_evidence_supports_every_verdict,
|
||||
check_intent_independence,
|
||||
check_isolation_was_examined,
|
||||
check_no_actor_collected_judgment,
|
||||
check_runs_agree,
|
||||
check_verdicts_follow_from_judgment_evidence,
|
||||
|
|
@ -56,6 +57,14 @@ def test_actor_isolation():
|
|||
assert check_actor_isolation(pack, SECRETS, memories) == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("mutations", [(), ("M15",), ("M01",)])
|
||||
def test_every_run_examines_isolation(mutations):
|
||||
"""F-0003 resolved: isolation is now a property of runs, not of scenarios
|
||||
written to expose it. No canaries need be planted by the test."""
|
||||
pack, _, _ = run_and_serialize(*mutations)
|
||||
assert check_isolation_was_examined(pack) == []
|
||||
|
||||
|
||||
# --- td://self/oracle-independence ---------------------------------------
|
||||
|
||||
|
||||
|
|
|
|||
136
tests/test_audit_core_e2_use_case.py
Normal file
136
tests/test_audit_core_e2_use_case.py
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
"""The audit-core use case is durable intent, even before drivers can run it."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from copy import deepcopy
|
||||
|
||||
from testdriver import Oracle, Provenance, Verdict
|
||||
from usecases.audit_core_e2_tenant_boundary import (
|
||||
PHASE_CONTRACTS,
|
||||
PRECEDENT_EVIDENCE_REF,
|
||||
ROLE_CONTRACTS,
|
||||
TEST_USE_CASE,
|
||||
)
|
||||
|
||||
|
||||
def passing_observations() -> dict[str, object]:
|
||||
absent = {
|
||||
"status": 404,
|
||||
"schema": ("$", "$.error:str"),
|
||||
"digest": "absent-surface",
|
||||
"fixture_match_count": 0,
|
||||
}
|
||||
return {
|
||||
"event_by_id": {
|
||||
"owner": {"status": 200, "fixture_match_count": 2},
|
||||
"attacker": dict(absent),
|
||||
"absent": dict(absent),
|
||||
},
|
||||
"correlation_slice": {
|
||||
"owner": {"status": 200, "fixture_match_count": 2},
|
||||
"attacker": {"status": 200, "fixture_match_count": 0},
|
||||
},
|
||||
"append_as_b": {
|
||||
"attacker": {"status": 400},
|
||||
"state_after": dict(absent),
|
||||
"absent": dict(absent),
|
||||
},
|
||||
"run": {
|
||||
"attempted_operations": 10,
|
||||
"maximum_operations": 30,
|
||||
"max_concurrency": 1,
|
||||
"rate_per_minute": 10,
|
||||
"target_revision_matches": True,
|
||||
"attribution_complete": True,
|
||||
},
|
||||
"cleanup": {
|
||||
"completed_before_expiry": True,
|
||||
"runner_absent": True,
|
||||
"temporary_identities": [],
|
||||
"kv_paths": [],
|
||||
"projection_resources": [],
|
||||
"mounted_secrets": [],
|
||||
"target_ready": True,
|
||||
"secret_values_observed": False,
|
||||
},
|
||||
"report": {
|
||||
"bound_to_projection_receipt": True,
|
||||
"bound_to_cleanup_receipt": True,
|
||||
"bounded_assurance_statement": True,
|
||||
"delivered_to_risk_nexus": True,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def judgments(observations):
|
||||
oracle = Oracle()
|
||||
assertions = (*TEST_USE_CASE.claims, *TEST_USE_CASE.invariants)
|
||||
return {
|
||||
assertion.id: oracle.judge(
|
||||
assertion,
|
||||
observations,
|
||||
getattr(assertion, "after_step", None),
|
||||
)
|
||||
for assertion in assertions
|
||||
}
|
||||
|
||||
|
||||
def test_precedent_calibration_satisfies_every_assertion():
|
||||
result = judgments(passing_observations())
|
||||
assert {judgment.verdict for judgment in result.values()} == {Verdict.PASS}
|
||||
assert PRECEDENT_EVIDENCE_REF.endswith(
|
||||
"AUDIT-WP-0008-T05-whitehat-e2-03-pass-2026-08-22.md"
|
||||
)
|
||||
|
||||
|
||||
def test_cross_tenant_event_disclosure_fails_the_claim():
|
||||
observations = passing_observations()
|
||||
observations["event_by_id"]["attacker"] = {
|
||||
"status": 200,
|
||||
"schema": ("$", "$.event_id:str"),
|
||||
"digest": "tenant-b-event",
|
||||
"fixture_match_count": 2,
|
||||
}
|
||||
result = judgments(observations)
|
||||
assert result["c-event-id-tenant-scoped"].verdict is Verdict.FAIL
|
||||
|
||||
|
||||
def test_missing_cleanup_evidence_is_inconclusive():
|
||||
observations = passing_observations()
|
||||
del observations["cleanup"]
|
||||
result = judgments(observations)
|
||||
assert result["c-receipt-bound-cleanup"].verdict is Verdict.INCONCLUSIVE
|
||||
assert result["i-no-secret-evidence"].verdict is Verdict.INCONCLUSIVE
|
||||
|
||||
|
||||
def test_cleanup_residue_fails_closed():
|
||||
observations = deepcopy(passing_observations())
|
||||
observations["cleanup"]["temporary_identities"] = ["unexpected-residue"]
|
||||
result = judgments(observations)
|
||||
assert result["c-receipt-bound-cleanup"].verdict is Verdict.FAIL
|
||||
|
||||
|
||||
def test_intent_has_independent_provenance_and_separate_roles():
|
||||
assert TEST_USE_CASE.provenance is Provenance.SPEC
|
||||
assert all(
|
||||
assertion.provenance is Provenance.SPEC
|
||||
for assertion in (*TEST_USE_CASE.claims, *TEST_USE_CASE.invariants)
|
||||
)
|
||||
roles = {role.id for role in ROLE_CONTRACTS}
|
||||
assert {
|
||||
"authorizer",
|
||||
"target-owner",
|
||||
"credential-custodian",
|
||||
"security-coordinator",
|
||||
"cluster-executor",
|
||||
"tenant-a-attacker",
|
||||
"tenant-b-control",
|
||||
"independent-observer",
|
||||
} == roles
|
||||
|
||||
|
||||
def test_schedule_requires_cleanup_before_finalization():
|
||||
phases = {phase.id: phase for phase in PHASE_CONTRACTS}
|
||||
assert phases["finalize-and-deliver"].requires == ("cleanup-custody",)
|
||||
assert phases["cleanup-custody"].requires == ("delete-runner",)
|
||||
assert phases["run-probes"].requires == ("ready-runner",)
|
||||
|
|
@ -194,3 +194,98 @@ def test_safe_to_accept_is_a_closed_set():
|
|||
assert SAFE_TO_ACCEPT == {
|
||||
Classification.UNCHANGED, Classification.MECHANICAL_ADAPTATION,
|
||||
}
|
||||
|
||||
|
||||
# --- classifications that no lab mutation happens to produce ---------------
|
||||
#
|
||||
# Two outcomes were declared at T08 and exercised by nothing. Left that way they
|
||||
# are decoration: code that has never run is code nobody has checked. Rather than
|
||||
# delete meaningful outcomes or trust them untested, both are given a case.
|
||||
|
||||
|
||||
def test_intent_change_is_detected_when_the_claim_set_moves(baseline):
|
||||
"""`INTENT_CHANGED` is a fact about the recorded use case, not an inference.
|
||||
|
||||
It fires because a human edited what is being asserted — which is why it is
|
||||
detectable at all, where `SEMANTIC_CHANGE` was not (F-0006).
|
||||
"""
|
||||
import copy
|
||||
|
||||
altered = copy.deepcopy(baseline)
|
||||
altered["provenance_index"]["c-newly-added-claim"] = "human"
|
||||
outcome = classify(baseline, altered)
|
||||
assert outcome.classification is Classification.INTENT_CHANGED
|
||||
assert not outcome.safe_to_accept
|
||||
|
||||
|
||||
def test_realization_failure_is_distinguishable_from_ambiguity():
|
||||
"""`REALIZATION_FAILED` says "we could not act"; `AMBIGUOUS` says "we do not know".
|
||||
|
||||
Every lab mutation that breaks realization also strands a claim, so the
|
||||
catalogue only ever produces `AMBIGUOUS`. This builds the case the catalogue
|
||||
cannot: a step that fails while every assertion in the run still holds and
|
||||
none of them depended on it.
|
||||
|
||||
Note that a run asserting *nothing at all* is `AMBIGUOUS`, not
|
||||
`REALIZATION_FAILED` — a use case with no claims cannot conclude anything,
|
||||
however well its steps ran.
|
||||
"""
|
||||
from testdriver import (
|
||||
Actor, Cast, Invariant, Oracle, Runner, Scenario, SemanticAction,
|
||||
StateObserver, Step, UseCase, VerificationAsset, World,
|
||||
)
|
||||
from testdriver.agentic import DiscoveryRuntime
|
||||
from testdriver.browser import BrowserDriver
|
||||
from testdriver.observers import Watch
|
||||
from testdriver.provenance import Provenance
|
||||
from lab.mutations import ObservationChannel
|
||||
|
||||
use_case = UseCase(
|
||||
"uc-audit-only", "Sharing leaves an ordered audit trail",
|
||||
"Alice shares R with Bob; the audit trail stays ordered.",
|
||||
Provenance.HUMAN,
|
||||
invariants=(Invariant(
|
||||
"i-audit-ordered", "The audit trail is append-only", Provenance.HUMAN,
|
||||
lambda obs: [e["sequence"] for e in obs["audit:R"]]
|
||||
== sorted(e["sequence"] for e in obs["audit:R"]),
|
||||
),),
|
||||
)
|
||||
|
||||
def run(*mutations):
|
||||
with journey_lab_server(*mutations) as (app, tokens, base_url):
|
||||
app.request(tokens["alice"], "create_resource",
|
||||
resource_id="R", content="x")
|
||||
cast = Cast()
|
||||
cast.add(Actor("alice", "Alice", credentials={"token": tokens["alice"]}))
|
||||
scenario = Scenario(
|
||||
"sc-audit-only", use_case,
|
||||
watches=(Watch("bob", "R"),),
|
||||
steps=(Step("s1", "alice", SemanticAction(
|
||||
"grant_access", {"subject_id": "bob", "permission": "READ"},
|
||||
permitted_surfaces=frozenset({"browser"}),
|
||||
)),),
|
||||
)
|
||||
driver = BrowserDriver(base_url, tokens, DiscoveryRuntime(), "R")
|
||||
observer = StateObserver(ObservationChannel(app), scenario.watches)
|
||||
world = World("w-audit", app, app.version, cast=cast)
|
||||
return json.loads(
|
||||
Runner(world, driver, observer, Oracle())
|
||||
.run(VerificationAsset("va-audit-only", scenario))
|
||||
.evidence.to_json()
|
||||
)
|
||||
|
||||
outcome = classify(run(), run("M23")) # the control is gone from the UI
|
||||
assert outcome.classification is Classification.REALIZATION_FAILED
|
||||
assert not outcome.safe_to_accept
|
||||
|
||||
|
||||
def test_no_classification_is_unreachable():
|
||||
"""Every declared outcome must be produced somewhere in this suite.
|
||||
|
||||
An outcome nothing can emit is the same kind of dead promise `SUSPICIOUS`
|
||||
was before T10 removed it.
|
||||
"""
|
||||
exercised = set(EXPECTED.values()) | {
|
||||
Classification.INTENT_CHANGED, Classification.REALIZATION_FAILED,
|
||||
}
|
||||
assert exercised == set(Classification)
|
||||
|
|
|
|||
|
|
@ -67,4 +67,18 @@ def test_actors_hold_isolated_credentials_and_memory():
|
|||
assert alice.credentials["token"] != bob.credentials["token"]
|
||||
alice.remember("secret", "only alice knows this")
|
||||
assert bob.recall("secret") is None
|
||||
assert bob.known_keys() == ()
|
||||
# Every actor carries its own automatic private marker (F-0003) and nothing
|
||||
# else it was not given.
|
||||
assert bob.known_keys() == ("__canary__",)
|
||||
assert alice.canary != bob.canary
|
||||
|
||||
|
||||
def test_every_run_records_a_verdict_on_isolation():
|
||||
"""F-0003: isolation is examined on every scenario, not only on ones
|
||||
written to expose it."""
|
||||
result, _ = run_once()
|
||||
examined = [
|
||||
obs for obs in result.evidence.observations if obs.kind == "actor_isolation"
|
||||
]
|
||||
assert len(examined) == 1
|
||||
assert examined[0].data["violations"] == []
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue