From ce353adde82917658a4d0ccb90e4fae46cac830d Mon Sep 17 00:00:00 2001 From: tegwick Date: Sat, 1 Aug 2026 13:39:12 +0200 Subject: [PATCH] =?UTF-8?q?CB-WP-0006=20T08:=20control=20loop=20=E2=80=94?= =?UTF-8?q?=204=20of=2014=20to=208=20of=2014,=20and=20a=20cost=20regressio?= =?UTF-8?q?n?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Test 1: the enforced count rose. AM-2, AM-6, AM-9 and AM-11 moved from unmutatable to red; AM-7's hash clause was re-earned so it is 2/3 rather than 1/3. Kernel spec->code link 15/18 -> 18/18, names only. Both denominators are stated. 8 of 14 is 57%, but four rows cannot be enforced — AM-3 blocked on an artifact, AM-4c withdrawn, AM-5 declared ungated by the spec, AM-10 withdrawn — so it is 8 of 10 enforceable. The 14 stays the headline and AM-4c stays in it on purpose: a score improved by deleting the question is not an improvement. Test 2: one row regressed and was caught. Moving AM-6's gate from debug to release turned its mutation SURVIVED, because 4,000 black_box iterations were calibrated against debug's 3.4x headroom and are invisible against release's 20x. The generalizable finding is that a weak mutation is not a fixed property of a row — it can become weak when the row's measurement conditions change, without the row, the mutation or the code being touched. Final SURVIVED count: 0. Test 3 is now mechanical rather than asserted. mutation-check gained an EXPECT-VACUOUS verdict: if a row's expect string appears in PASSING output, the FA guard would accept any failure at all, so the row is reported broken rather than red. Final run: 0 vacuous expects across 14 rows. The control exists because the failure happened — my first expect for AM-2 was "AM-2", which appears in the passing report and would have accepted a compile error as proof of enforcement. The cost result is a refutation, not a win. Mechanical share rose to 50%, the highest ever recorded and above the 38% baseline that motivated CB-WP-0004. That is not a tooling regression: environment setup and task closes are still at zero two passes on. It is the other half of CB-WP-0004 T06's finding arriving in force — text patching (45 turns, $13.79) and orientation (19 turns, $10.97) never had their manual path removed, and a code-heavy pass is exactly where that spends. Mean context 493,486 against a 200,000 target, up from 315,170. SessionShape has stated SS-01..SS-05 since CB-WP-0003 and none has ever been enforced — the only acceptance-adjacent numbers in this project with no gate at all, in a pass whose entire subject was ungated numbers. Numbering corrected: the workplan said CB-EV-0004, which CB-WP-0005 already used. This is CB-EV-0005. Co-Authored-By: Claude Opus 5 --- evidence/CB-EV-0005-instrument-the-table.md | 163 ++++++++++++++++++++ tools/mutation-check.py | 19 ++- 2 files changed, 178 insertions(+), 4 deletions(-) create mode 100644 evidence/CB-EV-0005-instrument-the-table.md diff --git a/evidence/CB-EV-0005-instrument-the-table.md b/evidence/CB-EV-0005-instrument-the-table.md new file mode 100644 index 0000000..e532523 --- /dev/null +++ b/evidence/CB-EV-0005-instrument-the-table.md @@ -0,0 +1,163 @@ +# CB-EV-0005: did instrumenting the acceptance table change what it enforces? + +research: [CB-RES-0004](../research/CB-RES-0004-replay-and-kernel-coverage.md) +adr: [ADR-0005](../decisions/ADR-0005-assertion-coverage-and-replay.md) +workplan: [CB-WP-0006](../workplans/CB-WP-0006-instrument-the-table.md) +log: [260801-cb-wp-0006-log.md](../history/260801-cb-wp-0006-log.md) +instruments: `make mutation-check`, `make coverage`, `make cost-mix` +window: `cb-cost --since dfd0d6d` — 144 responses, **$51.52** + +> **Numbering correction.** CB-WP-0006 T08 says "commit +> `evidence/CB-EV-0004`". That number was already taken by CB-WP-0005's +> evidence. This is **CB-EV-0005**. + +--- + +## Test 1 — did the enforced count rise, per row? + +**M-D1-MUT: 4 of 14 → 8 of 14.** + +| row | before | after | task | +|---|---|---|---| +| AM-1 | red | red | — | +| **AM-2** | unmutatable | **red** | T02 | +| AM-3 | unmutatable | unmutatable — **blocked**, artifact never built | T02 | +| AM-4a / AM-4b | red | red | — | +| AM-4c | unmutatable | unmutatable — **withdrawn**, retained in denominator | T04 | +| AM-5 | unmutatable | unmutatable — **measured and met**, spec declares it ungated | T03 | +| **AM-6** | unmutatable | **red** | T01 | +| AM-7 | PARTIAL 1/3 | PARTIAL **2/3** — `hash-identical` re-earned | T06 | +| AM-8 | PARTIAL 1/2 | PARTIAL 1/2 | — | +| **AM-9** | unmutatable | **red** | T03 | +| AM-10 | unmutatable | unmutatable — withdrawn | — | +| **AM-11** | unmutatable | **red** | T05 | +| AM-12 | red | red | — | + +Per-task predictions, reported against what happened: + +| task | predicted | measured | +|---|---|---| +| T01 AM-6 | `unmutatable` → `red` | **met** | +| T02 AM-2 | instrumented | **met** | +| T02 AM-3 | instrumented **or** restated | **restated** — blocked on an artifact, not a metric | +| T03 AM-9 | measured | **met** | +| T03 AM-5 | measured **or** withdrawn | **measured**, and the first reading was wrong (below) | +| T04 AM-4c | targeted **or** dropped | **dropped**, with the argument | +| T05 AM-11 | `unmutatable` → `red` | **met** | +| T06 K10 | replay real, AM-7 re-earnable | **met** — hash clause re-earned | +| T07 K14/K18 | implement **or** amend | **one of each** | + +**The honest denominator, both ways.** 8 of 14 is 57%. But four rows +*cannot* be enforced — AM-3 is blocked on an artifact, AM-4c is withdrawn, +AM-5 is declared ungated by the spec, AM-10 is withdrawn — and two are +partial. Of the **10 enforceable** rows, **8 are enforced**. + +The 14 stays the headline. AM-4c is deliberately retained in that +denominator after withdrawal, and the tool says so on every run: *a score +improved by deleting the question is not an improvement.* + +**Kernel spec→code link: 15/18 → 18/18.** With the caveat the gate prints +every run: that is names, not assertions. + +## Test 2 — did any row regress from `red` to `SURVIVED`? + +**Yes, once, and it was caught.** + +In T04, moving AM-6's gate from debug to release turned its mutation +`SURVIVED`. The 4,000 `black_box` iterations were calibrated against +debug's 3.4× headroom and are invisible against release's 20×. Raised to +100,000; back to `red`. + +The generalizable finding: **a weak mutation is not a fixed property of a +row. It can become weak when the row's measurement conditions change** — +without the row, the mutation, or the code being touched. That is a +maintenance burden M-D1-MUT carries and T09 should weigh. + +No other row regressed. `SURVIVED` count across the final run: **0**. + +## Test 3 — were the new mutations strong? + +The workplan required every new mutation to be *shown to fail for the +stated reason*. That is now **mechanical, not asserted**. + +`mutation-check` gained an `EXPECT-VACUOUS` verdict: if a row's `expect` +string appears in **passing** output, the FA guard would accept any +failure at all, and the row is reported broken rather than `red`. + +**Final run: 0 vacuous expects across all 14 rows.** + +This control exists because the failure it detects happened. My first +`expect` for AM-2 was `"AM-2"` — which appears in the passing report, so +the guard would have accepted a compile error as proof the row was +enforced. Caught by hand in T02; caught mechanically from T08 onward. + +Every mutation added this pass is a **property** mutation, not a threshold +tweak: + +| row | mutation | expect | +|---|---|---| +| AM-6 | 100,000 `black_box` iterations in `GroundState::fold` | `AM-6 UNMET` | +| AM-2 | ~800 filler lines into the impl | `FAIL target <= 40` | +| AM-9 | a 300 MB allocation in the workload | `FAIL target <= 64 MB` | +| AM-11 | `NullRng::draw` returns its bound | `outside 0..` | + +## What the instruments caught that nothing else would + +Six defects, all found by a gate rather than by review: + +1. **AM-6's gate measured contention, not throughput** — 38,753 ev/s + inside `make all` against 341,280 in isolation, because `cargo test` + runs concurrently. +2. **AM-5's first reading was wrong by 61%** — 87.0 s under load against + 37.3 s quiet. Reported as a 45% breach; it is met with 1.6× headroom. + *Same error class as (1), committed two tasks later by the same author + in the row immediately after diagnosing it.* +3. **RSS over-reported 3×** — `getrusage(RUSAGE_CHILDREN)` is a + high-water mark across every reaped child, so `cargo`'s memory was + charged to the workload. Found only by cross-checking `/usr/bin/time`. +4. **K10's first round trip did not reproduce** — hashing a + `serde_json::Value` is a different canonical form than hashing the + typed aggregate. A round trip that recomputed its own comparison value + would have *passed* this. +5. **The bench workload existed twice** — `bench_shape` kept passing + while `bench-test` broke on the same edit. +6. **`facts-check` caught two spec copies going stale** within the hour + the numbers moved. + +## Cost, and a regression that matters + +| | CB-WP-0004 window | CB-WP-0005 | **CB-WP-0006** | +|---|---|---|---| +| pass cost | — | $16.03 | **$51.52** | +| responses | — | 83 | **144** | +| mechanical share | 26% | 38% | **50%** | +| environment setup | 3 turns | 0 | **0** | +| hub + workplan edits | 1 turn | 0 | **0** | +| ad-hoc text patching | 8 turns | 11 | **45 turns, $13.79** | +| orientation / inspect | 41 turns | 7 | **19 turns, $10.97** | +| mean context | — | 315,170 | **493,486** | + +**Mechanical share rose to 50% — the highest ever recorded, above the 38% +baseline that motivated CB-WP-0004.** This is not a tooling regression: +environment setup and task closes are still at **zero**, two passes on. +It is the *other* half of CB-WP-0004 T06's finding arriving in force — +text patching and orientation never had their manual path removed, and a +code-heavy pass is exactly where that spends. + +**Context is the sharper problem.** Mean 493,486 against a 200,000 target, +up from 315,170 last pass. `specs/SessionShape.md` has stated SS-01…SS-05 +since CB-WP-0003 and **none of them has ever been enforced** — the only +acceptance-adjacent numbers in this project with no gate at all, in a pass +whose entire subject was ungated numbers. + +## Verdict + +| claim | status | +|---|---| +| enforced count rose | **confirmed** — 4 → 8 of 14, 8 of 10 enforceable | +| kernel link complete | **confirmed** — 18/18, names only | +| no row regressed to SURVIVED | **confirmed** — one near-miss, caught | +| new mutations are strong | **confirmed mechanically** — 0 vacuous expects | +| AM-3 instrumented | **not met** — blocked on an artifact, stated not hidden | +| AM-7, AM-8 fully enforced | **not met** — partial, reported as partial | +| the pass got cheaper | **refuted** — mechanical share rose to 50% | diff --git a/tools/mutation-check.py b/tools/mutation-check.py index c3a578e..06682fd 100644 --- a/tools/mutation-check.py +++ b/tools/mutation-check.py @@ -282,10 +282,20 @@ def check_row(row): # Positive control 2: the baseline must be green, or "mutant red" # proves nothing. - base_ok, base_tail, _ = run(row.verify) + base_ok, base_tail, base_out = run(row.verify) if not base_ok: return "inconclusive", f"baseline already red: {base_tail}" + # CB-WP-0006 T08: the FA guard is only a guard if its `expect` string + # cannot appear in PASSING output. An expect of "AM-2" would match the + # normal report and accept any failure at all — which is how the guard + # goes vacuous without anyone noticing. My first attempt on AM-2 did + # exactly that. + if row.expect and row.expect in base_out: + return "EXPECT-VACUOUS", ( + f"expect string {row.expect!r} appears in PASSING output, so it " + f"would accept any failure — the FA guard is inert for this row") + try: mutated = original.replace(old, new, 1) # Positive control 3: the file content must actually differ. @@ -337,12 +347,13 @@ def report(only=None): detail = (f"{len(r.clauses) - len(unmet)}/{len(r.clauses)} " f"clauses enforced") tally[verdict] = tally.get(verdict, 0) + 1 - if verdict == "HARNESS-BROKEN": + if verdict in ("HARNESS-BROKEN", "EXPECT-VACUOUS"): broken.append(r.id) mark = {"red": "red ", "SURVIVED": "SURVIVED ", "unmutatable": "unmutatable", "inconclusive": "inconclusive", "PARTIAL": "PARTIAL ", "HARNESS-BROKEN": "BROKEN ", + "EXPECT-VACUOUS": "EXPECT-VOID", "WRONG-REASON": "WRONG-REASON"}[verdict] print(f" [{mark}] {r.id:<6} {r.claim[:52]}") if detail: @@ -366,8 +377,8 @@ def report(only=None): print(f" ({withdrawn} withdrawn row(s) retained in the denominator " f"on purpose — a score\n improved by deleting the question " f"is not an improvement)") - for k in ("PARTIAL", "SURVIVED", "WRONG-REASON", "unmutatable", - "inconclusive"): + for k in ("PARTIAL", "SURVIVED", "WRONG-REASON", "EXPECT-VACUOUS", + "unmutatable", "inconclusive"): if tally.get(k): print(f" {k:<13} {tally[k]}") if only: