test-driver/tests/test_lab_ground_truth.py
tegwick 84848e9a0e T08: the classifier, measured and attacked
False Adaptation Rate = 0/7 across the labelled catalogue and the three E-003
attacks. 11 of 12 mechanical mutations absorbed without a human, so the safety
result is not bought by escalating everything.

- classification.py: total function over three signals, rule order chosen so
  every rule that could excuse a regression sits after the rule that reports
  one. SAFE_TO_ACCEPT is a two-element closed set, asserted.
- CompositeDriver plus scenarios/full_journey.py: one asset crossing both
  surfaces, so UI mutations are visible as surface differences while the
  claims they do not touch stay green.
- E-003: surface substitution (new M23), concurrent mechanical+defect,
  evidence starvation, provenance laundering. All held.

F-0006 (CONCEPT_DRIFT, resolved): the T02 design listed SEMANTIC_CHANGE as an
outcome the table could produce. It cannot - M12 and M19 are behaviourally
identical, as the lab has asserted since T05. PRODUCT_DEFECT and
SEMANTIC_CHANGE collapse into one escalating outcome, BEHAVIOUR_CHANGED, and
the distinction becomes a human adjudication. INTENT_CHANGED survives but is
detected by the claim fingerprint moving, not inferred from behaviour.

Two classifier defects found and fixed rather than reported: claims downstream
of a failed realization now yield INCONCLUSIVE rather than FAIL (a false
accusation is the mirror image of a false adaptation), and the browser driver
records a page signature so surface change is detectable when the interaction
path is unchanged.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

Assistant: claude-code
Assistant-Model: opus
Assistant-Process: 1629012@bnt-lap001
Assistant-Session: 78d4fb13-8a1e-474b-87a3-9b9261c49a39
2026-08-23 00:02:58 +02:00

174 lines
7.2 KiB
Python

"""The lab is the measuring instrument. These tests keep it honest.
Nothing here tests the kernel's cleverness — the kernel has no classifier yet.
What is established is the **ground truth** every later measurement is taken
against: which mutations the reference scenario responds to, and how.
The response matrix below is an asserted fact, not a snapshot. If a change to the
lab or the scenario moves a cell, that is a change to the measuring instrument
and must be a deliberate, reviewed act.
"""
from __future__ import annotations
import pytest
from testdriver import Runner, Verdict
from lab.mutations import BY_ID, CATALOGUE, build_lab, expected_classification
from scenarios.alice_bob_carol import build
def run_against(*mutations: str):
world, driver, observer, asset, oracle = build(*mutations)
return Runner(world, driver, observer, oracle).run(asset)
# --- the catalogue itself -------------------------------------------------
def test_catalogue_is_large_enough_to_support_a_rate():
"""Six mutations cannot support precision or recall. Twenty can begin to."""
assert len(CATALOGUE) >= 20 # 23 as of T08
def test_every_mutation_is_labelled_and_reasoned():
for mutation in CATALOGUE:
assert mutation.label in ("MECHANICAL", "SEMANTIC", "DEFECT")
assert mutation.rationale.strip(), f"{mutation.id} has no rationale"
def test_labels_cover_all_three_classes_with_useful_weight():
counts = {label: 0 for label in ("MECHANICAL", "SEMANTIC", "DEFECT")}
for mutation in CATALOGUE:
counts[mutation.label] += 1
assert all(count >= 4 for count in counts.values()), counts
def test_every_mutation_is_reproducible_and_version_stamped():
for mutation in CATALOGUE:
first, _ = build_lab(mutation.id)
second, _ = build_lab(mutation.id)
assert first.version == second.version == f"lab-0.2.0-{mutation.id}"
assert first.applied_mutations == (mutation.id,)
def test_mutations_compose_and_record_both():
"""Row 3 of the decision table needs a lab carrying both at once."""
app, _ = build_lab("M01", "M15")
assert app.version == "lab-0.2.0-M01+M15"
assert app.applied_mutations == ("M01", "M15")
def test_test_id_axis_is_represented():
"""H-001 must be analysable split by whether stable selectors survived."""
preserved = [m.id for m in CATALOGUE if m.preserves_test_ids]
dropped = [m.id for m in CATALOGUE if not m.preserves_test_ids]
assert preserved and dropped, "both sides of the test-id axis must exist"
# --- the response matrix --------------------------------------------------
# Ground truth: what the reference scenario reports for each lab version.
# `None` means "no assertion in this scenario covers this mutation" — recorded
# honestly rather than papered over.
EXPECTED_VERDICT = {
"M01": Verdict.PASS, "M02": Verdict.PASS, "M03": Verdict.PASS,
"M04": Verdict.PASS, "M05": Verdict.PASS, "M06": Verdict.PASS,
"M07": Verdict.PASS, "M08": Verdict.PASS, "M09": Verdict.PASS,
"M10": Verdict.PASS,
"M11": Verdict.FAIL, "M12": Verdict.FAIL,
"M13": Verdict.PASS, "M14": Verdict.PASS,
"M15": Verdict.FAIL, "M16": Verdict.FAIL, "M17": Verdict.FAIL,
"M18": Verdict.FAIL, "M19": Verdict.FAIL, "M20": Verdict.FAIL,
}
# Mutations the reference scenario cannot see, and why. Coverage is scoped to
# what a scenario asserts (F-0002) *and* to the surfaces it touches: this
# scenario is API-only, so a defect that lives in the UI is outside its reach.
# Declaring them is mandatory — `test_out_of_scope_mutations_are_declared`
# fails on any invisible mutation that is not named here.
OUT_OF_SCOPE = {
"M13": "only affects grants that omit a permission; the scenario passes READ explicitly",
"M14": "a change of intent with no change of code; nothing observable moved",
"M21": "a UI mutation; this scenario never loads the UI",
"M22": "a UI mutation; this scenario never loads the UI",
"M23": "a UI-surface defect; this scenario is API-only. Covered by the "
"cross-surface journey in tests/test_classification.py, where it is "
"the E-003 surface-substitution attack.",
}
def test_baseline_passes():
assert run_against().verdict is Verdict.PASS
@pytest.mark.parametrize("mutation_id", sorted(EXPECTED_VERDICT))
def test_response_matrix_is_stable(mutation_id):
assert run_against(mutation_id).verdict is EXPECTED_VERDICT[mutation_id]
def test_no_mechanical_mutation_changes_the_verdict():
"""Semantics are preserved, so the use case must not notice."""
for mutation in CATALOGUE:
if mutation.label == "MECHANICAL":
assert run_against(mutation.id).verdict is Verdict.PASS, mutation.id
def test_every_in_scope_defect_is_detected():
"""The floor of the whole project. A defect the lab cannot surface is a
defect no later classifier can be measured against."""
missed = [
m.id for m in CATALOGUE
if m.label == "DEFECT"
and m.id not in OUT_OF_SCOPE
and run_against(m.id).verdict is Verdict.PASS
]
assert missed == [], f"undetected seeded defects: {missed}"
def test_out_of_scope_mutations_are_declared():
"""A mutation the scenario cannot see must be named, not silently ignored.
Applies to defects as much as to semantic changes — an undeclared invisible
defect is precisely how a suite comes to look greener than it is.
"""
for mutation in CATALOGUE:
if run_against(mutation.id).verdict is not Verdict.PASS:
continue
if mutation.label == "MECHANICAL":
continue # passing is the correct outcome for these
assert mutation.id in OUT_OF_SCOPE, (
f"{mutation.id} ({mutation.label}) is invisible to the reference "
"scenario and undeclared — either cover it or record why not"
)
def test_declared_out_of_scope_mutations_really_are_invisible():
"""Stale declarations rot silently. If a mutation becomes visible, the
declaration must be removed rather than left as a standing excuse."""
for mutation_id in OUT_OF_SCOPE:
assert run_against(mutation_id).verdict is Verdict.PASS, (
f"{mutation_id} is declared out of scope but the reference scenario "
"now detects it — remove the declaration"
)
def test_deferred_revoke_and_revoke_race_are_behaviourally_identical():
"""M12 (SEMANTIC) and M19 (DEFECT) must be indistinguishable from evidence.
This is the discrimination problem in its sharpest form, and the reason
classification cannot be a diff over observed behaviour. Both produce the
same failure; only intent separates them, which is why claims need
independent provenance and why ambiguity escalates to a human.
"""
semantic = run_against("M12")
defect = run_against("M19")
failed = lambda r: sorted({j.assertion_id for j in r.judgments
if j.verdict is not Verdict.PASS})
assert failed(semantic) == failed(defect) == ["c-bob-revoked"]
assert expected_classification("M12") != expected_classification("M19")
def test_a_mechanical_change_shipping_with_a_defect_still_fails():
"""Decision-table row 3: coincidence is not exoneration."""
assert run_against("M01", "M15").verdict is Verdict.FAIL