clay-borg/tools/rule-coverage.py
tegwick 9e30bbb2b3 T09: close the spec->code link mechanically (M-D1-LNK)
make coverage now reports a second number: how many claimed rules are
also NAMED in the aggregate source. 58/58 tag coverage was weaker
evidence than it read as, and this says how much weaker.

Measured: 49 of 58. Nine rules are claimed by a scenario and appear
nowhere in games/ground/src/lib.rs --
GR-D07 GR-F02 GR-L03 GR-O03 GR-P01 GR-P02 GR-P03 GR-P04 GR-T01.

The gate REPORTS rather than fails, on purpose. Closing the gap by
adding those IDs to comments would satisfy the check without
establishing that any of the nine is implemented -- the overclaim
InnerLoop implementation rule 2 exists to prevent, and one CB-WP-0001
already committed once. Each needs its implementation confirmed before
it is tagged; promoting M-D1-LNK to a failing gate is correct after
that, not before.

Also added: a phantom check that fails when a rule id appears in code
that the spec does not define (currently zero).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-07-31 09:27:14 +02:00

141 lines
5 KiB
Python
Executable file

#!/usr/bin/env python3
"""AM-1 / M-D1-COV: every numbered GR-rule needs >=1 scenario.
Compares the rule IDs declared in specs/GroundRules.md against the
`covers:` lists in scenarios/ground/*.yaml. Exits non-zero when a
scenario claims a rule the spec does not define, so coverage can never
be inflated by a typo'd or invented rule ID.
Stated limit (InnerLoop implementation rule 4): this gate counts tags. It
proves no rule is unclaimed and no claimed rule is invented. It does NOT
prove a scenario exercises the rule it names.
Positive control (InnerLoop v1.1 §Step 5): the run asserts it actually
found rules and scenarios. Before this was added, a broken spec regex
yielded rules=[] and missing=[] and the tool exited 0 reporting "0/0"
the harness-does-nothing class, in the tool that reports our headline
coverage number.
Usage:
python3 tools/rule-coverage.py
python3 tools/rule-coverage.py --self-test
"""
import glob
import re
import sys
RULE_RE = r"\*\*(GR-[A-Z]+\d+)"
COVERS_RE = r"covers: \[(.*?)\]"
AGGREGATE = "games/ground/src/lib.rs"
ID_RE = r"GR-[A-Z]+\d+"
def parse_rules(spec_text):
return sorted(set(re.findall(RULE_RE, spec_text)))
def parse_code_ids(text):
"""Rule IDs named anywhere in the aggregate source (T09)."""
return set(re.findall(ID_RE, text))
def parse_covers(text):
match = re.search(COVERS_RE, text, re.S)
if not match:
return set()
return {c.strip() for c in match.group(1).split(",") if c.strip()}
def self_test():
"""Each assertion pins a failure this tool must detect."""
results = []
def check(name, ok, detail=""):
results.append((name, ok, detail))
# The defect that motivated this control: a spec that parses to zero
# rules must not be reportable as coverage.
check("zero rules detected as a failure", parse_rules("no rules here") == [],
"empty spec yields no rules; main() now aborts on this")
# The matcher must actually match the real format.
check("rule matcher works on real spec format",
parse_rules("**GR-R06** something\n**GR-A12** other")
== ["GR-A12", "GR-R06"])
# covers: parsing, including the empty case.
check("covers matcher works", parse_covers("covers: [GR-R06, GR-A12]")
== {"GR-R06", "GR-A12"})
check("missing covers yields empty set", parse_covers("no covers key") == set())
# T09: the spec -> code link must be detectable.
check("code-id matcher finds ids in source",
parse_code_ids("// GR-R06: lead first\nfn f(){} // GR-A12")
== {"GR-R06", "GR-A12"})
check("code-id matcher finds none in unmarked source",
parse_code_ids("fn f() { let x = 1; }") == set())
print("rule-coverage self-test (positive control)")
ok = True
for name, passed, detail in results:
print(f" [{'ok ' if passed else 'FAIL'}] {name}"
+ (f"{detail}" if detail else ""))
ok &= passed
return 0 if ok else 1
def main():
if "--self-test" in sys.argv:
return self_test()
rules = parse_rules(open("specs/GroundRules.md").read())
paths = sorted(glob.glob("scenarios/ground/*.yaml"))
# Positive control: refuse to report a percentage over nothing.
if not rules:
print("ERROR — no GR-rules parsed from specs/GroundRules.md; "
"refusing to report coverage", file=sys.stderr)
return 1
if not paths:
print("ERROR — no scenarios found in scenarios/ground/; "
"refusing to report coverage", file=sys.stderr)
return 1
covered = set()
for path in paths:
covered |= parse_covers(open(path).read())
known = set(rules)
hit = sorted(known & covered)
missing = [r for r in rules if r not in covered]
invented = sorted(covered - known)
# T09: the spec -> code -> scenario chain, made mechanical. A rule a
# scenario claims should also be named in the aggregate, or the claim
# rests on nothing but a tag.
code_ids = parse_code_ids(open(AGGREGATE).read())
unlinked = sorted((known & covered) - code_ids)
phantom = sorted(code_ids - known)
pct = 100 * len(hit) // len(rules)
linked = len((known & covered) & code_ids)
print(f"AM-1 rule coverage: {len(hit)}/{len(rules)} ({pct}%) "
f"over {len(paths)} scenarios")
print(f"AM-1b spec->code link: {linked}/{len(hit)} claimed rules also "
f"named in {AGGREGATE}")
print(" NOTE: counts tags; does not prove a scenario exercises what it names")
if unlinked:
print(" unlinked (claimed by a scenario, absent from the aggregate):")
print(" ", " ".join(unlinked))
if phantom:
print(" ERROR — rule id in code that the spec does not define:",
" ".join(phantom), file=sys.stderr)
return 1
if missing:
print(" uncovered:", " ".join(missing))
if invented:
print(" ERROR — claimed but not defined in the spec:", " ".join(invented),
file=sys.stderr)
return 1
return 0 if not missing else 2
if __name__ == "__main__":
sys.exit(main())