diff --git a/Makefile b/Makefile index 89a7113..d786a54 100644 --- a/Makefile +++ b/Makefile @@ -24,7 +24,7 @@ TOOLS := $(REPO)/tools # Every cargo recipe runs at the repo root; the shell does not persist cd. IN_REPO := cd $(REPO) && -.PHONY: check test sim bench bench-test coverage dep-weight cost cost-test cost-pin cost-budget cost-mix loop-lint self-tests env-test task-done status loc all +.PHONY: check test sim bench bench-test coverage dep-weight cost cost-test cost-pin cost-budget cost-mix loop-lint self-tests env-test task-done status facts-check facts-gen loc all ## fmt + clippy (deny warnings) + HashMap deny-lint check: @@ -63,6 +63,7 @@ self-tests: $(PY) $(TOOLS)/repo.py --self-test $(PY) $(TOOLS)/task-done.py --self-test $(PY) $(TOOLS)/status.py --self-test + $(PY) $(TOOLS)/facts.py --self-test # T01 positive control: prove the environment fix, do not assume it. Runs # every tool from a foreign working directory with a PATH that has no @@ -81,6 +82,15 @@ env-test: @$(MAKE) -C $(REPO) coverage >/dev/null \ && echo " [ok ] make -C works from any directory" +# T04: single source of fact (InnerLoop v1.2) — the DFD gate. +# facts.toml is GENERATED; facts-check fails if it disagrees with the +# instruments, or if a tagged artifact disagrees with it. +facts-check: + $(PY) $(TOOLS)/facts.py --check + +facts-gen: + $(PY) $(TOOLS)/facts.py --gen + # T03: one-shot orientation — workplans, next task, spend, fast gates. # Cheap by design: no build. Start a session with this instead of grepping. status: @@ -124,4 +134,4 @@ loc: printf '%-28s %s\n' $$d "$$(find $$d/src -name '*.rs' | xargs cat | grep -vcE '^\s*(//|$$)')"; \ done -all: check test sim coverage dep-weight self-tests env-test loop-lint bench-test +all: check test sim coverage dep-weight self-tests env-test facts-check loop-lint bench-test diff --git a/decisions/ADR-0004-am4-ratification.md b/decisions/ADR-0004-am4-ratification.md index 1223b9a..fde8be8 100644 --- a/decisions/ADR-0004-am4-ratification.md +++ b/decisions/ADR-0004-am4-ratification.md @@ -49,8 +49,8 @@ At the time of the retarget, `make dep-weight`: | configuration | crates | third-party LOC | |---|---|---| -| shipped-runtime (`--no-default-features`) | 23 | **246,250** | -| dev-toolchain (default features) | 29 | **317,021** | +| shipped-runtime (`--no-default-features`) | 23 | **246,250** | +| dev-toolchain (default features) | 29 | **317,021** | | own source | — | 3,443 | ## Why these targets bind on future work rather than merely passing present work diff --git a/evidence/CB-EV-0001-game-kernel.md b/evidence/CB-EV-0001-game-kernel.md index 8bc157f..2b923a8 100644 --- a/evidence/CB-EV-0001-game-kernel.md +++ b/evidence/CB-EV-0001-game-kernel.md @@ -17,8 +17,8 @@ Machine: WSL2, Linux 6.18.33.2-microsoft-standard-WSL2, rustc 1.97.1, | Metric | Target | Measured | Verdict | |---|---|---|---| | AM-1 rule coverage | 100% of GR-rules | 58/58 (100%) | **met** | -| AM-4a dep weight, shipped runtime | ≤250,000 third-party lines | 246,250 (23 crates) | **met** | -| AM-4b dep weight, dev toolchain | ≤350,000 third-party lines | 317,021 (29 crates) | **met** | +| AM-4a dep weight, shipped runtime | ≤250,000 third-party lines | 246,250 (23 crates) | **met** | +| AM-4b dep weight, dev toolchain | ≤350,000 third-party lines | 317,021 (29 crates) | **met** | | AM-6 throughput | ≥100,000 events/s | 1,651,400 events/s | **met, 16.5×** | | AM-7 scaling | ≥0.9× at 20× workload | 1.08× | **met** | | AM-7 replay | 100k events ≤5s | 2.18 ms (CI 2.14–2.23) | **met, 2,290×** | @@ -152,8 +152,8 @@ retargeted onto third-party source under audit. | Configuration | Crates | Third-party LOC | Target | Verdict | |---|---|---|---|---| -| Shipped runtime (`--no-default-features`) | 23 | 246,250 | ≤250,000 | **met** | -| Dev toolchain (default features) | 29 | 317,021 | ≤350,000 | **met** | +| Shipped runtime (`--no-default-features`) | 23 | 246,250 | ≤250,000 | **met** | +| Dev toolchain (default features) | 29 | 317,021 | ≤350,000 | **met** | | Our own source | — | 3,408 | — | — | Scenario tooling costs **70,771 lines that a shipped game never diff --git a/evidence/CB-EV-0002-cost-accounting.md b/evidence/CB-EV-0002-cost-accounting.md index e193449..4800d4c 100644 --- a/evidence/CB-EV-0002-cost-accounting.md +++ b/evidence/CB-EV-0002-cost-accounting.md @@ -17,7 +17,7 @@ transcribed by hand (CA-15). | ID | Metric | Target | Measured | Verdict | |---|---|---|---|---| -| AC-1 | pinned total, as two components | $93.15 = $92.03 + $1.11 | **$92.03 main + $1.11 subagent = $93.15** | **met** | +| AC-1 | pinned total, as two components | $93.15 = $92.03 + $1.11 | **$92.03 main + $1.11 subagent = $93.15** | **met** | | AC-2 | reconciliation residual | $0.00 | **$0.000000** | **met** | | AC-3 | unattributed share reported | present, 33% | **32.4%, own line** | **met** | | AC-4 | composition reported | 5 components | **5 of 5** | **met** | diff --git a/facts.toml b/facts.toml new file mode 100644 index 0000000..4081f65 --- /dev/null +++ b/facts.toml @@ -0,0 +1,100 @@ +# GENERATED — do not edit. `make facts-gen` rewrites this file. +# +# The single source of fact for numbers that appear in more than +# one artifact (InnerLoop v1.2). Every value here is produced by +# the instrument named in its `by` field, on the current tree. +# `make facts-check` fails if this file disagrees with the +# instruments, or if a tagged artifact disagrees with this file. + +generated = "2026-07-31" +pin = "fc76445" + +[am4a_loc] +value = 246250 +text = "246,250" +fmt = "{:,}" +by = "tools/dep-weight.py" + +[am4a_target] +value = 250000 +text = "250,000" +fmt = "{:,}" +by = "tools/dep-weight.py TARGETS" + +[am4b_loc] +value = 317021 +text = "317,021" +fmt = "{:,}" +by = "tools/dep-weight.py" + +[am4b_target] +value = 350000 +text = "350,000" +fmt = "{:,}" +by = "tools/dep-weight.py TARGETS" + +[gr_covered] +value = 58 +text = "58" +fmt = "{:,}" +by = "tools/rule-coverage.py" + +[gr_linked] +value = 49 +text = "49" +fmt = "{:,}" +by = "tools/rule-coverage.py" + +[gr_rules] +value = 58 +text = "58" +fmt = "{:,}" +by = "tools/rule-coverage.py" + +[gr_scenarios] +value = 21 +text = "21" +fmt = "{:,}" +by = "tools/rule-coverage.py" + +[pinned_main] +value = 92.03371920000004 +text = "$92.03" +fmt = "${:,.2f}" +by = "tools/cb-cost.py --pin fc76445" + +[pinned_mechanical_cost] +value = 33.3506319 +text = "$33.35" +fmt = "${:,.2f}" +by = "tools/cb-cost.py --pin fc76445" + +[pinned_mechanical_share] +value = 36 +text = "36%" +fmt = "{:d}%" +by = "tools/cb-cost.py --pin fc76445" + +[pinned_mechanical_turns] +value = 167 +text = "167" +fmt = "{:,}" +by = "tools/cb-cost.py --pin fc76445" + +[pinned_responses] +value = 346 +text = "346" +fmt = "{:,}" +by = "tools/cb-cost.py --pin fc76445" + +[pinned_subagent] +value = 1.1137235 +text = "$1.11" +fmt = "${:,.2f}" +by = "tools/cb-cost.py --pin fc76445" + +[pinned_total] +value = 93.14744270000004 +text = "$93.15" +fmt = "${:,.2f}" +by = "tools/cb-cost.py --pin fc76445" diff --git a/specs/CostAccounting.md b/specs/CostAccounting.md index aa19d7a..1cb3cfc 100644 --- a/specs/CostAccounting.md +++ b/specs/CostAccounting.md @@ -135,7 +135,7 @@ Each row names the command that produces its number, per InnerLoop §Step 4. | ID | Metric | Target | Instrument | |---|---|---|---| -| **AC-1** | reproduces the pinned CB-WP-0001 total | **$93.15** = $92.03 main + $1.11 subagent | `make cost-pin` | +| **AC-1** | reproduces the pinned CB-WP-0001 total | **$93.15** = $92.03 main + $1.11 subagent | `make cost-pin` | | **AC-2** | reconciliation residual (CA-14) | **$0.00** exactly | same command, `reconciled: ok` line | | **AC-3** | unattributed share reported (CA-10) | present, and **32.4%** on the pinned run | `cb-cost --pin fc76445 --by-task` | | **AC-4** | composition reported (CA-13) | all five components present | `cb-cost --pin fc76445 --composition` | @@ -155,7 +155,7 @@ assertion against a fixture whose expected value is known and fails loudly; Per InnerLoop §Step 4, the acceptance table is checked against the contracts in this same spec: -- AC-1's $93.15 is reachable only if CA-06 holds (both trees enumerated). +- AC-1's $93.15 is reachable only if CA-06 holds (both trees enumerated). Under a main-file-only collector the target is unreachable — this is the defect the adversarial review caught, where a target of $92.21 would have been hit *only* by a broken collector. diff --git a/specs/InnerLoop.md b/specs/InnerLoop.md index 3e1ccee..b3e505e 100644 --- a/specs/InnerLoop.md +++ b/specs/InnerLoop.md @@ -1,7 +1,11 @@ # The Inner Loop — Assimilate and Surpass -Status: **v1.2** — corrected from CB-WP-0003 (loop hardening) on -2026-07-31. Changes from v1.1: single source of fact; review targets the +Status: **v1.3** — corrected from CB-WP-0004 (mechanical work) on +2026-07-31. Change from v1.2: single source of fact is now executable +(`make facts-check`, CB-WP-0004 T04), giving the duplicated-fact-drift +class its first gate. + +v1.2 changes from v1.1: single source of fact; review targets the harness and states its sampling limit; correction vs retarget; the chaos roll's calibration window; the live cost budget. The design goal is now stated: **optimize for cheap correction, not for exhaustive prevention.** @@ -39,7 +43,16 @@ command so re-running is free. > acceptance figure had to be chased across three artifacts each time it > moved. No positive control catches this — both copies are internally > consistent — and re-derivation does not either, because the copy -> reproduces whatever it was copied from.)* The loop's own optimization target is **agentic efficiency**: +> reproduces whatever it was copied from.)* +> +> **Now executable (v1.3, CB-WP-0004 T04):** `make facts-check`. `facts.toml` +> is generated from the instruments, never edited; an artifact quoting a +> registry value tags it `` and the gate fails when the +> two disagree. Untagged literal copies are reported, not failed — that is +> the drift surface still uncovered, and naming it is more useful than +> pretending it is closed. + +The loop's own optimization target is **agentic efficiency**: every artifact it produces must be small enough to load whole, structured enough to act on without interpretation, and falsifiable enough that an agent can judge its own work without a human in the iteration. diff --git a/specs/MetricsAndScenarios.md b/specs/MetricsAndScenarios.md index 9ae2846..9d523f8 100644 --- a/specs/MetricsAndScenarios.md +++ b/specs/MetricsAndScenarios.md @@ -56,7 +56,7 @@ rule 4). a rule a scenario claims should also appear in the aggregate source, or the claim rests on a tag and nothing else. -Measured 2026-07-31: **49 of 58** claimed rules are named in +Measured 2026-07-31: **49 of 58** claimed rules are named in `games/ground/src/lib.rs`. **Unmet**, target 58. The nine unlinked: ```text diff --git a/tools/__pycache__/dep-weight.cpython-312.pyc b/tools/__pycache__/dep-weight.cpython-312.pyc new file mode 100644 index 0000000..912597b Binary files /dev/null and b/tools/__pycache__/dep-weight.cpython-312.pyc differ diff --git a/tools/__pycache__/rule-coverage.cpython-312.pyc b/tools/__pycache__/rule-coverage.cpython-312.pyc new file mode 100644 index 0000000..a921196 Binary files /dev/null and b/tools/__pycache__/rule-coverage.cpython-312.pyc differ diff --git a/tools/facts.py b/tools/facts.py new file mode 100644 index 0000000..00dd687 --- /dev/null +++ b/tools/facts.py @@ -0,0 +1,369 @@ +#!/usr/bin/env python3 +"""The fact registry: a number lives in one place, and copies are checked. + +CB-WP-0004 T04. Duplicated-fact drift (DFD) is the fourth error class on +record and the only one with no executable gate. Two instances: + + * `specs/MetricsAndScenarios.md` §1a inlined a copy of the price sheet; + the real sheet changed and the copy went stale within the hour. + * the acceptance figure moved $92.21 -> $92.87 -> $93.32 -> $93.15 and + each move had to be chased by hand across a survey, a workplan and an + evidence file. + +No positive control catches DFD — both copies are internally consistent — +and re-derivation does not either, because the copy reproduces whatever it +was copied from. It is caught only by reading a copy against its source. +That is what this does. + +## The trap, and how it is avoided + +A hand-maintained registry is just another copy that drifts. So the +registry is **generated from the instruments** (`make facts-gen`), never +edited: every value here is produced by cb-cost, dep-weight or +rule-coverage on the current tree. `--check` re-runs the instruments and +fails if the committed registry disagrees with them, so a stale registry +cannot silently certify stale artifacts. + +## How an artifact quotes a fact + +Tag the occurrence with an HTML comment naming the registry key: + + The benchmark to beat is **$93.15**. + +`make facts-check` re-reads every tagged occurrence and fails when the +text disagrees with the registry. Tagging is opt-in, so the gate also +reports **untagged copies** — literal occurrences of a registry value in +artifacts that did not declare them — which is the drift surface that is +not yet covered. + +Usage: + python3 tools/facts.py --gen # regenerate facts.toml from instruments + python3 tools/facts.py --check # gate: registry vs instruments vs artifacts + python3 tools/facts.py --self-test +""" +import datetime +import glob +import importlib.util +import os +import re +import subprocess +import sys + +from repo import ROOT, enter_root + +REGISTRY = os.path.join(ROOT, "facts.toml") +PIN = "fc76445" # CB-WP-0001 acceptance pin (specs/CostAccounting.md §7) +TAG_RE = re.compile(r"") +# Artifact kinds that quote measured numbers. `history/` is excluded from +# the untagged sweep: it narrates corrections and legitimately states +# superseded values. +ARTIFACT_DIRS = ("specs", "evidence", "research", "decisions", "workplans") + +try: + import tomllib +except ModuleNotFoundError: # pragma: no cover + print("ERROR: needs Python 3.11+ for tomllib", file=sys.stderr) + sys.exit(1) + + +def _load(script): + spec = importlib.util.spec_from_file_location( + script.replace("-", "_").replace(".py", ""), + os.path.join(ROOT, "tools", script)) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +# --------------------------------------------------------------- generate + + +def measure(): + """Every registry value, produced by the instrument that owns it. + + Each entry is (value, format, instrument). The format string is how the + value appears in prose, so the check compares like with like rather + than re-deriving a rendering in two places. + """ + facts = {} + + cb = _load("cb-cost.py") + rep = cb.collect("-home-worsch-clay-borg", PIN) + facts["pinned_total"] = (rep["total"], "${:,.2f}", + f"tools/cb-cost.py --pin {PIN}") + facts["pinned_main"] = (rep["main_total"], "${:,.2f}", + f"tools/cb-cost.py --pin {PIN}") + facts["pinned_subagent"] = (rep["subagent_total"], "${:,.2f}", + f"tools/cb-cost.py --pin {PIN}") + facts["pinned_responses"] = (rep["responses"], "{:,}", + f"tools/cb-cost.py --pin {PIN}") + mix = rep["tool_mix"] + facts["pinned_mechanical_cost"] = (mix["mechanical_cost"], "${:,.2f}", + f"tools/cb-cost.py --pin {PIN}") + facts["pinned_mechanical_turns"] = (mix["mechanical_turns"], "{:,}", + f"tools/cb-cost.py --pin {PIN}") + facts["pinned_mechanical_share"] = ( + round(100 * mix["mechanical_cost"] / rep["total"]), "{:d}%", + f"tools/cb-cost.py --pin {PIN}") + + dw = _load("dep-weight.py") + for label, key in (("shipped-runtime", "am4a_loc"), ("dev-toolchain", "am4b_loc")): + found = dw.crates(dw.CONFIGS[label]) + total = sum(dw.source_lines(n, v) for n, v in sorted(found.items())) + facts[key] = (total, "{:,}", "tools/dep-weight.py") + facts["am4a_target"] = (dw.TARGETS["shipped-runtime"], "{:,}", + "tools/dep-weight.py TARGETS") + facts["am4b_target"] = (dw.TARGETS["dev-toolchain"], "{:,}", + "tools/dep-weight.py TARGETS") + + rc = _load("rule-coverage.py") + rules = rc.parse_rules(open(os.path.join(ROOT, "specs/GroundRules.md")).read()) + paths = sorted(glob.glob(os.path.join(ROOT, "scenarios/ground/*.yaml"))) + covered = set() + for p in paths: + covered |= rc.parse_covers(open(p).read()) + code_ids = rc.parse_code_ids(open(os.path.join(ROOT, rc.AGGREGATE)).read()) + hit = set(rules) & covered + facts["gr_rules"] = (len(rules), "{:,}", "tools/rule-coverage.py") + facts["gr_covered"] = (len(hit), "{:,}", "tools/rule-coverage.py") + facts["gr_linked"] = (len(hit & code_ids), "{:,}", "tools/rule-coverage.py") + facts["gr_scenarios"] = (len(paths), "{:,}", "tools/rule-coverage.py") + return facts + + +def render(value, fmt): + return fmt.format(value) + + +def generate(): + facts = measure() + lines = [ + "# GENERATED — do not edit. `make facts-gen` rewrites this file.", + "#", + "# The single source of fact for numbers that appear in more than", + "# one artifact (InnerLoop v1.2). Every value here is produced by", + "# the instrument named in its `by` field, on the current tree.", + "# `make facts-check` fails if this file disagrees with the", + "# instruments, or if a tagged artifact disagrees with this file.", + "", + f'generated = "{datetime.date.today().isoformat()}"', + f'pin = "{PIN}"', + "", + ] + for key in sorted(facts): + value, fmt, by = facts[key] + lines += [ + f"[{key}]", + f"value = {value!r}", + f'text = "{render(value, fmt)}"', + f'fmt = "{fmt}"', + f'by = "{by}"', + "", + ] + with open(REGISTRY, "w") as fh: + fh.write("\n".join(lines)) + print(f"wrote {os.path.relpath(REGISTRY, ROOT)} — {len(facts)} facts") + for key in sorted(facts): + print(f" {key:<26} {facts[key][1].format(facts[key][0]):>12}" + f" {facts[key][2]}") + return 0 + + +# ------------------------------------------------------------------ check + + +def load_registry(): + if not os.path.isfile(REGISTRY): + return None + with open(REGISTRY, "rb") as fh: + return tomllib.load(fh) + + +def artifacts(): + for d in ARTIFACT_DIRS: + base = os.path.join(ROOT, d) + if not os.path.isdir(base): + continue + for path in sorted(glob.glob(os.path.join(base, "**", "*.md"), + recursive=True)): + yield os.path.relpath(path, ROOT) + + +def tagged_occurrences(text): + """(key, line_no, the text of that line) for each fact tag.""" + out = [] + for i, line in enumerate(text.splitlines(), 1): + for m in TAG_RE.finditer(line): + out.append((m.group(1), i, line)) + return out + + +def check(): + reg = load_registry() + if reg is None: + print(f"ERROR — {os.path.relpath(REGISTRY, ROOT)} missing; " + f"run `make facts-gen`", file=sys.stderr) + return 1 + keys = {k: v for k, v in reg.items() if isinstance(v, dict)} + + findings = [] + + # 1. The registry must still agree with the instruments. Without this + # a stale registry would happily certify stale artifacts — the + # trap this task was warned about. + live = measure() + for key, (value, fmt, _by) in sorted(live.items()): + if key not in keys: + findings.append(f"registry missing {key} (instrument says " + f"{render(value, fmt)}) — run `make facts-gen`") + continue + if render(value, fmt) != keys[key]["text"]: + findings.append( + f"registry stale: {key} = {keys[key]['text']} but " + f"{keys[key]['by']} now measures {render(value, fmt)} " + f"— run `make facts-gen`") + for key in sorted(set(keys) - set(live)): + findings.append(f"registry has {key}, no instrument produces it " + f"— hand-edited?") + + # 2. Every tagged occurrence must state the registry value. + tagged = 0 + for rel in artifacts(): + text = open(os.path.join(ROOT, rel)).read() + for key, lineno, line in tagged_occurrences(text): + tagged += 1 + if key not in keys: + findings.append(f"{rel}:{lineno} tags unknown fact {key!r}") + continue + if keys[key]["text"] not in line: + findings.append( + f"{rel}:{lineno} claims fact:{key} but does not state " + f"{keys[key]['text']} | {line.strip()[:70]}") + + # 3. Positive control. A gate that checked nothing would pass silently + # — the harness-does-nothing class, in the tool meant to close DFD. + if tagged == 0: + print("ERROR — no fact tags found in any artifact; the check verified " + "nothing. Tag at least one occurrence, or delete this gate.", + file=sys.stderr) + return 1 + + # 4. Report untagged copies: the drift surface still uncovered. This + # reports and does not fail — a number can legitimately recur (a + # round figure, a year), and failing on that would make the gate + # something people route around. + untagged = {} + for rel in artifacts(): + text = open(os.path.join(ROOT, rel)).read() + declared = {k for k, _, _ in tagged_occurrences(text)} + for i, line in enumerate(text.splitlines(), 1): + if TAG_RE.search(line): + continue + for key, spec in keys.items(): + if key in declared: + continue + # Only distinctive values: a bare "58" is everywhere. + if len(spec["text"]) < 5: + continue + if spec["text"] in line: + untagged.setdefault(key, []).append(f"{rel}:{i}") + + print("facts-check — single source of fact (DFD gate)") + print(f" registry {len(keys)} facts, generated {reg.get('generated')}, " + f"pin {reg.get('pin')}") + print(f" tagged {tagged} occurrence(s) checked against the registry") + if untagged: + n = sum(len(v) for v in untagged.values()) + print(f" untagged {n} literal copy/copies, not covered by the gate:") + for key in sorted(untagged): + print(f" {key:<26} {' '.join(untagged[key][:6])}" + + (" …" if len(untagged[key]) > 6 else "")) + print(" NOTE: reported, not failed — tag them to bring them under " + "the gate") + + if findings: + print(f"\n{len(findings)} finding(s):", file=sys.stderr) + for f in findings: + print(f" [facts] {f}", file=sys.stderr) + return 1 + print(" no findings") + return 0 + + +# --------------------------------------------------------------- self-test + + +def self_test(): + """Each assertion pins a failure this gate must detect. + + The class being closed is DFD, so the assertions are about *copies*: + a copy that disagrees must fail, a copy that agrees must pass, and a + run that checked no copies at all must not report success. + """ + results = [] + + def check_(name, ok, detail=""): + results.append((name, ok, detail)) + + check_("tag is recognised in prose", + tagged_occurrences("cost was **$93.15** ") + == [("pinned_total", 1, "cost was **$93.15** ")]) + check_("untagged prose yields no occurrence", + tagged_occurrences("cost was $93.15") == []) + check_("tag with spaces is recognised", + [k for k, _, _ in tagged_occurrences("x ")] + == ["am4a_loc"]) + check_("a malformed tag is not silently accepted", + tagged_occurrences("") == []) + + # The DFD detection itself: agreeing and disagreeing copies. + spec = {"text": "$93.15"} + agree = "the benchmark is **$93.15** " + drift = "the benchmark is **$92.87** " + check_("an agreeing copy passes", spec["text"] in agree) + check_("a drifted copy is detected", spec["text"] not in drift, + "this is the whole class: both lines are internally consistent") + + # Registry must exist, be complete, and match the instruments. + reg = load_registry() + check_("registry exists", reg is not None) + if reg: + keys = {k: v for k, v in reg.items() if isinstance(v, dict)} + check_("registry is non-empty", len(keys) >= 8, f"{len(keys)} facts") + check_("every fact records the instrument that produced it", + all(v.get("by") for v in keys.values())) + check_("every fact records both a value and its rendered text", + all("value" in v and v.get("text") for v in keys.values())) + check_("registry is generated, not hand-written", + open(REGISTRY).read().startswith("# GENERATED")) + + # At least one artifact must actually be tagged, or the gate is inert. + total = sum(len(tagged_occurrences(open(os.path.join(ROOT, r)).read())) + for r in artifacts()) + check_("at least one artifact is tagged (gate is not inert)", total > 0, + f"{total} tagged occurrence(s)") + + print("facts self-test (positive control)") + ok = True + for name, passed, det in results: + print(f" [{'ok ' if passed else 'FAIL'}] {name}" + + (f" — {det}" if det else "")) + ok &= passed + return 0 if ok else 1 + + +def main(): + enter_root() + if "--self-test" in sys.argv: + return self_test() + if "--gen" in sys.argv: + return generate() + return check() + + +if __name__ == "__main__": + try: + sys.exit(main()) + except subprocess.CalledProcessError as e: + print(f"ERROR — {e}", file=sys.stderr) + sys.exit(1) diff --git a/workplans/CB-WP-0004-mechanical-work.md b/workplans/CB-WP-0004-mechanical-work.md index affa88a..03a9eac 100644 --- a/workplans/CB-WP-0004-mechanical-work.md +++ b/workplans/CB-WP-0004-mechanical-work.md @@ -136,7 +136,7 @@ believed because it was produced by a program rather than by hand. ```task id: CB-WP-0004-T03 -status: todo +status: done priority: medium state_hub_task_id: "5466a510-37a5-4491-b93e-509cd400cc23" ``` @@ -207,6 +207,34 @@ and say so — a gate with no generator still closes the class. **Predicted:** **$6–9** recovered, plus DFD's first executable gate. Confidence medium; this is the hardest task here and the most valuable. +**Delivered — both halves, not just the check.** `facts.toml` holds 15 +facts and is **generated** by `make facts-gen` from cb-cost, dep-weight +and rule-coverage; the file opens with `# GENERATED — do not edit` and +the self-test asserts that line is still there. The trap named in this +task — a hand-maintained registry that becomes another drifting copy — +is closed by `facts-check` re-running the instruments and failing if the +committed registry disagrees with them. A stale registry cannot certify +stale artifacts. + +An artifact quoting a fact tags it: `**$93.15** `. +17 occurrences across 5 artifacts are now under the gate. + +**Falsified before being believed.** Changing `specs/CostAccounting.md` +line 158 from $93.15 to $92.87 — the exact historical drift — produced +exit 1 and `specs/CostAccounting.md:158 claims fact:pinned_total but does +not state $93.15`. The gate was tested against the class it exists to +catch, on a real artifact, not only in its self-test. + +**What it does not close, stated rather than implied.** 22 untagged +literal copies remain, across `specs/InnerLoop.md`, `specs/GameKernel.md`, +`research/CB-RES-0002` and the older workplans. They are **reported, not +failed**: tagging is opt-in, a number can legitimately recur, and a gate +that fires on coincidence gets routed around. Naming the uncovered +surface is more useful than claiming the class is closed. + +InnerLoop's single-source-of-fact rule moves from prose to executable — +**v1.3**. + ## Phase C — Prove it, or withdraw the claim ## Task: Control loop — measure recovery and test for relocation