Some checks failed
ci / check (push) Has been cancelled
CB-EV-0009. Tier S did not produce a worse outcome than tier L would have. The roll deleted a survey that would have opened on 2D toolkit selection; the decomposition it forced instead found that the existing text renderer was showing 24 of 41 view fields. The structural trigger fires on a property of the plan, not of the code, so nothing in the tier derivation could have said the port was the wrong first question. Recorded honestly in both directions: the pass also made an interface change with no review, which is the cost side. One favourable fire is not a calibration; the window stays open to 2026-09-30. Corrects a number this pass asserted: the T01 commit message says '42 of 43, up from 24'. Measured by splicing the old renderer back in, it is 41 of 42 with 1 declared omitted, up from 24 -- and 16 fields were genuinely absent, not 17, because outcome.winners was rendered in a different format. Both original figures were counted by hand. CHAOS gains its first caught entry. Cheapest pass per response yet recorded (0.094 vs a previous best of 0.123).
136 lines
5.3 KiB
TOML
136 lines
5.3 KiB
TOML
# The gate registry (ADR-0006 D3, CB-WP-0009 T02).
|
|
#
|
|
# A gate without an expiry is a permanent tax justified once. Every
|
|
# standing control mechanism gets an entry here saying what it checks,
|
|
# what it has actually **caught**, when its keep-or-kill argument is due,
|
|
# and what would retire it.
|
|
#
|
|
# `make gate-review` reports what is overdue and what has caught nothing.
|
|
# It reports; it does not fail the build — CB-RES-0005 §4: a gate that
|
|
# blocks the remedy when the metric breaches is a trap, not a gate.
|
|
#
|
|
# `caught` is the load-bearing field. An empty `caught` is not proof a
|
|
# gate is useless — it may be preventing rather than missing — but it
|
|
# means the argument has to be made out loud on `review_by`.
|
|
|
|
# Targets in `make all` that are build or acceptance steps rather than
|
|
# *control* gates — they measure the product, not how we work. Listed
|
|
# explicitly so a new target has to be classified rather than ignored;
|
|
# `loop-lint` fails when a target is in neither list.
|
|
not_control_gates = [
|
|
"check", "test", "sim", "bench-test", "size-metrics", "runtime-metrics",
|
|
"am6", "replay-test", "dep-weight", "self-tests", "env-test",
|
|
]
|
|
|
|
[[gate]]
|
|
id = "CB-01/CB-02"
|
|
name = "cost budget"
|
|
target = "cost-budget"
|
|
checks = "spend since the last commit; soft $10, hard $22"
|
|
added = "2026-07-30"
|
|
review_by = "2026-11-30"
|
|
caught = [
|
|
"CB-WP-0005: hard breach forced the Phase C re-plan",
|
|
"CB-WP-0006 T07: $12.06 in one task, the pass's most expensive",
|
|
]
|
|
retire_if = "two consecutive passes never approach the soft line, or commits get small enough that the window is always trivial"
|
|
|
|
[[gate]]
|
|
id = "SH-1/SH-2/SH-3"
|
|
name = "session-shape budget"
|
|
target = "shape-budget"
|
|
checks = "mean and p90 context, and batching rate, since the last commit"
|
|
added = "2026-08-01"
|
|
review_by = "2026-11-30"
|
|
caught = [
|
|
"first run fired HARD at 656,574 against a 300,000 ceiling, which is what prompted the compaction before CB-WP-0008",
|
|
]
|
|
retire_if = "context stops correlating with cost, or the model's context handling makes the number unactionable"
|
|
|
|
[[gate]]
|
|
id = "M-D1-MUT"
|
|
name = "mutation coverage of acceptance rows"
|
|
target = "mutation-check"
|
|
checks = "each acceptance row's assertion must go red for a stated reason when mutated"
|
|
added = "2026-07-31"
|
|
review_by = "2026-12-31"
|
|
caught = [
|
|
"AM-6 measuring contention, not throughput",
|
|
"AM-5's 61% measurement error under load",
|
|
"peak RSS over-reported 3x",
|
|
"K10's first round trip not reproducing",
|
|
"the bench workload existing twice",
|
|
"AM-2's expect matching its own passing output (EXPECT-VACUOUS)",
|
|
]
|
|
retire_if = "a full pass adds rows without finding anything, twice running — the harness costs real money per run"
|
|
|
|
[[gate]]
|
|
id = "DFD"
|
|
name = "single source of fact"
|
|
target = "facts-check"
|
|
checks = "every tagged number in the docs matches the tool that measures it"
|
|
added = "2026-07-31"
|
|
review_by = "2026-12-31"
|
|
caught = [
|
|
"gr_scenarios stale at 21 after CB-WP-0008 T03 added three scenarios",
|
|
]
|
|
retire_if = "the untagged-literal count reaches zero and stays there, meaning the docs stopped restating measured numbers"
|
|
|
|
[[gate]]
|
|
id = "AM-1b"
|
|
name = "kernel spec->code link"
|
|
target = "coverage"
|
|
checks = "every numbered K-rule is named in the source; binds 2026-08-31"
|
|
added = "2026-07-31"
|
|
review_by = "2026-08-31"
|
|
caught = [
|
|
"3 unlinked K-rules at introduction (15/18); 18/18 today",
|
|
]
|
|
retire_if = "it stays at 100% through two passes that add kernel rules — at that point it is measuring a habit, not enforcing one"
|
|
|
|
[[gate]]
|
|
id = "META-25"
|
|
name = "meta budget"
|
|
target = "status"
|
|
checks = "share of the trailing 3 passes spent on the loop itself; soft 25%"
|
|
added = "2026-08-01"
|
|
review_by = "2026-11-30"
|
|
caught = [
|
|
"its own cumulative-window defect, reported in CB-EV-0007 §3 and fixed by CB-WP-0009 T01",
|
|
]
|
|
retire_if = "product and meta stop being separable, or the share sits under the line for four passes without anyone consulting it"
|
|
|
|
[[gate]]
|
|
id = "LOOP-LINT"
|
|
name = "executable InnerLoop rules"
|
|
target = "loop-lint"
|
|
checks = "loadability, unmeasured verdicts, tier and chaos declarations, review trails, self-test entry points, and this registry"
|
|
added = "2026-07-30"
|
|
review_by = "2026-12-31"
|
|
caught = [
|
|
"four loadability breaches (401, 427, 406, 409 lines), each fixed structurally rather than by raising the limit",
|
|
"a reporting tool with no --self-test entry point (tools/repo.py)",
|
|
]
|
|
retire_if = "two passes run with no finding while artifacts keep growing — that would mean it is measuring the wrong properties"
|
|
|
|
[[gate]]
|
|
id = "CHAOS"
|
|
name = "the chaos roll"
|
|
target = ""
|
|
checks = "d4 on each tier declaration, 12-declaration calibration window"
|
|
added = "2026-07-30"
|
|
review_by = "2026-09-30"
|
|
caught = [
|
|
"CB-WP-0011: first fire in 6 declarations — d4=4 rolled stage 1 from structural L to S; the deleted survey would have opened on 2D toolkits while the existing text renderer was showing 24 of 41 view fields (CB-EV-0009 §1)",
|
|
]
|
|
retire_if = "the window closes with no overridden tier producing a different outcome than the argued one — the evaluation this window exists to make possible"
|
|
|
|
[[gate]]
|
|
id = "GATE-REVIEW"
|
|
name = "this registry"
|
|
target = "gate-review"
|
|
checks = "gates past their review date, and gates that have caught nothing"
|
|
added = "2026-08-01"
|
|
review_by = "2026-12-31"
|
|
caught = []
|
|
retire_if = "it has retired, tightened, or forced the re-justification of nothing by its review date — then it is a ritual, and ADR-0006 D4 says rituals cash out or go"
|