clay-borg/tools/loop-lint.py
tegwick 69c1658d20
Some checks failed
ci / check (push) Has been cancelled
CB-WP-0019 T03/T04: the cost rule written down, and the lifecycle
T03: InnerLoop v1.7 plus loop-lint's own-cost check. Six passes
under-reported themselves by 30-45%, never once high, and the rule lived
only in evidence files having been re-derived three times. The READING
is load bearing, not the boundary: CB-WP-0018 T04 applied 're-run the
instrument at the moment of quoting' alone and its figure was correct.
So the operative instruction is re-run when you quote, and loop-lint
fails an evidence file naming its own workplan beside a dollar amount
without marking it provisional.

It binds forward from this pass. The check fires on seven historical
files which ARE the evidence for the rule; making them comply would edit
the record to remove the thing it proves -- the same category error as a
live fact: tag on a dated measurement, which this pass also hit.

Lifecycle, at the maintainer's instruction: ready -> active -> done,
where ready means declared and not started. loop-lint fails a workplan
that has started and still says ready, one that is active with
everything closed, and one that is done with an open task. The first
version of that check was WRONG and its own self-test caught it: it
stripped the leading status: assuming frontmatter, which silently
dropped a real task once the frontmatter said ready or active.

Both new checks then fired on this pass's own artifacts and both were
right to.

T04: CB-EV-0017. The new meta budget's first reading is a breach it
caused -- 27% against the 20% line, because this pass cost $31.18
against product passes averaging ~$21. Reported rather than exempted:
ADR-0006 D2 covers the instrument repairs but not the rule-writing, and
the honest reading is that this should have been two passes.

CB-WP-0018 settled at $36.53/95 against $28.08/82 last reported, 30%
higher. Seven for seven.

make all exits 0.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-03 19:25:18 +02:00

483 lines
19 KiB
Python

#!/usr/bin/env python3
"""Executable checks for specs/InnerLoop.md rules (CB-WP-0003 T01).
The loop's own rules were prose. A rule nobody can run is a suggestion,
and the audit in history/260731-inner-loop-rule-audit.md found several
that were already being violated with no signal. This makes the
mechanically-checkable ones fail a command.
Each check names the InnerLoop rule it enforces. Checks that cannot be
made mechanical are recorded in the audit as `checkable` or `decorative`
and are deliberately absent here — see the audit for why.
Positive control (InnerLoop v1.1 §Step 5): --self-test asserts each check
actually detects its failure, using fixtures with known answers. A linter
that passes everything because its matcher is broken is the same defect
class as a benchmark timing rejected work.
Usage:
python3 tools/loop-lint.py # lint the repo
python3 tools/loop-lint.py --self-test # positive control
"""
import os
import re
import sys
from repo import ROOT as REPO # noqa: E402 (single source of fact, T01)
LOADABILITY_LIMIT = 400
# Artifact classes the loop produces. history/ is an append-only trail
# (verbatim challenge text is not something to split), so it is exempt.
LOOP_DIRS = ("specs", "research", "decisions", "evidence", "workplans")
class Finding:
def __init__(self, rule, path, detail):
self.rule, self.path, self.detail = rule, path, detail
def __str__(self):
return f" [{self.rule}] {self.path}\n {self.detail}"
def _md_files(root=REPO):
"""Loop artifacts only.
Vendored third-party trees (a baseline harness ships its own
node_modules) are not artifacts this loop produces, and linting them
buries the two real findings under eighteen irrelevant ones.
"""
for d in LOOP_DIRS:
base = os.path.join(root, d)
for dirpath, dirnames, files in os.walk(base):
dirnames[:] = [x for x in dirnames if x != "node_modules"]
for f in sorted(files):
if f.endswith(".md"):
yield os.path.relpath(os.path.join(dirpath, f), root)
def check_loadability(root=REPO):
"""§Agentic-efficiency 1 — every loop artifact stays under ~400 lines."""
out = []
for rel in _md_files(root):
with open(os.path.join(root, rel)) as fh:
n = sum(1 for _ in fh)
if n > LOADABILITY_LIMIT:
out.append(
Finding(
"loadability",
rel,
f"{n} lines exceeds the ~{LOADABILITY_LIMIT}-line limit; "
f"split and link with relative paths",
)
)
return out
def check_evidence_no_unmeasured(root=REPO):
"""§Rubric — `unmeasured` is legal in a survey, illegal in an evidence file."""
out = []
base = os.path.join(root, "evidence")
if not os.path.isdir(base):
return out
for f in sorted(os.listdir(base)):
if not f.endswith(".md"):
continue
rel = os.path.join("evidence", f)
for i, line in enumerate(open(os.path.join(root, rel)), 1):
# A row asserting the verdict, not prose discussing the word.
if re.search(r"\|\s*unmeasured\s*\|", line):
out.append(
Finding("evidence-unmeasured", f"{rel}:{i}",
"verdict `unmeasured` in an evidence table")
)
return out
def check_own_cost_not_quoted(root=REPO):
"""§Quoting a cost — an evidence file may not quote its own pass's
cost as final.
Six passes under-reported themselves by 30-45%, never once high, so a
self-quoted figure is not a rounding error but a known bias. The check
is deliberately narrow: it fires only when a line names the file's OWN
workplan beside a dollar amount and does not mark it provisional.
Quoting an earlier pass is exactly what the rule asks for.
**Binds forward, from the pass that wrote it down.** The rule was
written in CB-WP-0019 after six passes had each under-reported
themselves, and those six evidence files are the *evidence for the
rule*. Firing on them would demand the record be edited to remove the
thing it proves — the same category error as putting a live `fact:`
tag on a dated measurement. So the check applies from CB-WP-0019 on,
and the older figures stay as they were reported.
"""
BINDS_FROM = 19
out = []
base = os.path.join(root, "evidence")
if not os.path.isdir(base):
return out
for f in sorted(os.listdir(base)):
if not f.endswith(".md"):
continue
rel = os.path.join("evidence", f)
text = open(os.path.join(root, rel)).read()
# The pass an evidence file belongs to is named in its header.
m = re.search(r"\b(CB-WP-\d{4})\b", text)
if not m:
continue
own = m.group(1)
if int(own.rsplit("-", 1)[1]) < BINDS_FROM:
continue
for i, line in enumerate(text.splitlines(), 1):
if own not in line or "$" not in line:
continue
if "provisional" in line.lower() or "not quoted" in line.lower():
continue
out.append(
Finding("own-cost", f"{rel}:{i}",
f"quotes {own}'s own cost as final; re-run the "
f"instrument and quote a settled pass, or mark it "
f"provisional")
)
return out
def check_workplan_lifecycle(root=REPO):
"""§Workplan lifecycle — `ready` means declared and NOT started.
A workplan with work done in it that still says `ready` answers the
wrong question: the status should say whether anyone is on it, not
only whether it is finished.
"""
out = []
base = os.path.join(root, "workplans")
if not os.path.isdir(base):
return out
for f in sorted(os.listdir(base)):
if not f.endswith(".md"):
continue
rel = os.path.join("workplans", f)
text = open(os.path.join(root, rel)).read()
m = re.search(r"^status:\s*(\S+)", text, re.M)
if not m:
continue
status = m.group(1)
# Parse the ```task blocks, not every `status:` in the file. The
# first version stripped the leading match assuming it was the
# frontmatter — which silently dropped a real task the moment the
# frontmatter said `ready` or `active`, because those do not match
# the task vocabulary. Caught by this check's own self-test.
tasks = [
t.group(1)
for block in re.findall(r"```task\n(.*?)```", text, re.S)
for t in [re.search(r"^status:\s*(\S+)", block, re.M)]
if t
]
if not tasks:
continue
closed = sum(1 for t in tasks if t in ("done", "cancel"))
started = closed > 0
if status == "ready" and started:
out.append(Finding("lifecycle", rel,
"still `ready` but work has started — use `active`"))
elif status == "active" and closed == len(tasks):
out.append(Finding("lifecycle", rel,
"every task is closed but status is `active` — use `done`"))
elif status == "done" and closed != len(tasks):
out.append(Finding("lifecycle", rel,
f"`done` with {len(tasks) - closed} task(s) still open"))
return out
def check_survey_tier_and_chaos(root=REPO):
"""§Loop tiers — tier declared, and the chaos roll recorded every time."""
out = []
base = os.path.join(root, "research")
if not os.path.isdir(base):
return out
for f in sorted(os.listdir(base)):
if not f.endswith(".md"):
continue
rel = os.path.join("research", f)
text = open(os.path.join(root, rel)).read()
if not re.search(r"^tier:\s*[SML]\b", text, re.M):
out.append(Finding("tier-declared", rel, "no `tier:` declaration"))
elif "chaos" not in text.lower():
out.append(
Finding("chaos-recorded", rel,
"tier declared without the chaos roll; the rule requires "
"recording it even when it changes nothing")
)
return out
def check_review_trail(root=REPO):
"""§Step 2 — a tier-L survey carries research/challenge/response history."""
out = []
base = os.path.join(root, "research")
hist = os.path.join(root, "history")
if not (os.path.isdir(base) and os.path.isdir(hist)):
return out
files = os.listdir(hist)
for f in sorted(os.listdir(base)):
if not f.endswith(".md"):
continue
rel = os.path.join("research", f)
text = open(os.path.join(root, rel)).read()
if not re.search(r"^tier:\s*L\b", text, re.M):
continue
if not re.search(r"^status:\s*approved", text, re.M):
continue
for kind in ("challenge", "response"):
if not any(x.endswith(f"-{kind}.md") and kind in x for x in files):
out.append(
Finding("review-trail", rel,
f"tier-L approved survey with no history/*-{kind}.md")
)
return out
def check_reporting_tools_self_test(root=REPO):
"""§Step 5 v1.1 — every tool that reports a number exposes --self-test."""
out = []
base = os.path.join(root, "tools")
if not os.path.isdir(base):
return out
for f in sorted(os.listdir(base)):
if not f.endswith(".py"):
continue
rel = os.path.join("tools", f)
text = open(os.path.join(root, rel)).read()
if "--self-test" not in text:
out.append(
Finding("self-test", rel,
"reporting tool with no --self-test entry point; "
"nothing verifies its positive control still works")
)
return out
def check_gate_registry(root=REPO):
"""ADR-0006 D3 — every control gate is in `gates.toml`, and every
entry names a real target.
The failure this prevents is drift in the direction nobody notices: a
gate added to `make all` with no registry entry never acquires a
review date, which is how five mechanisms accumulated with no way to
retire any of them.
"""
out = []
registry = os.path.join(root, "gates.toml")
makefile = os.path.join(root, "Makefile")
if not (os.path.exists(registry) and os.path.exists(makefile)):
return out
try:
import tomllib
except ModuleNotFoundError: # pragma: no cover
return out
with open(registry, "rb") as fh:
data = tomllib.load(fh)
gates = data.get("gate") or []
if not gates:
return [Finding("gates", "gates.toml", "registry contains no gates")]
registered = {g.get("target") for g in gates if g.get("target")}
exempt = set(data.get("not_control_gates") or [])
text = open(makefile).read()
m = re.search(r"^all:(.*)$", text, re.M)
deps = m.group(1).split() if m else []
targets = {ln.split(":", 1)[0].strip() for ln in text.splitlines()
if ln and not ln[0].isspace() and ":" in ln and not ln.startswith(".")}
for dep in deps:
if dep not in registered and dep not in exempt:
out.append(Finding(
"gates", "gates.toml",
f"`make all` runs {dep!r}, which is neither a registered "
f"control gate nor listed in not_control_gates — classify it, "
f"so it cannot acquire permanence without a review date"))
for target in sorted(registered):
if target not in targets:
out.append(Finding(
"gates", "gates.toml",
f"entry names target {target!r}, which the Makefile lacks"))
return out
CHECKS = (
check_loadability,
check_evidence_no_unmeasured,
check_own_cost_not_quoted,
check_workplan_lifecycle,
check_survey_tier_and_chaos,
check_review_trail,
check_reporting_tools_self_test,
check_gate_registry,
)
def self_test():
"""Each check must DETECT its failure, not merely run."""
import shutil
import tempfile
results = []
def check(name, ok, detail=""):
results.append((name, ok, detail))
tmp = tempfile.mkdtemp()
try:
for d in LOOP_DIRS + ("tools", "history"):
os.makedirs(os.path.join(tmp, d), exist_ok=True)
# loadability: 401 lines must trip, 400 must not.
with open(os.path.join(tmp, "specs", "Big.md"), "w") as fh:
fh.write("x\n" * (LOADABILITY_LIMIT + 1))
with open(os.path.join(tmp, "specs", "Ok.md"), "w") as fh:
fh.write("x\n" * LOADABILITY_LIMIT)
f = check_loadability(tmp)
check("loadability detects overlong artifact",
len(f) == 1 and "Big.md" in f[0].path,
f"{len(f)} finding(s)")
# own-cost: a file quoting its OWN pass beside a dollar amount
# trips; the same line marked provisional does not; and a file
# quoting an EARLIER pass does not, because that is the rule.
def ev(name, body):
with open(os.path.join(tmp, "evidence", name), "w") as fh:
fh.write(body)
ev("CB-EV-0100-self.md", "CB-WP-0019 T04.\n| CB-WP-0019 | $9.99 |\n")
f = check_own_cost_not_quoted(tmp)
check("own-cost detects a pass quoting itself", len(f) == 1,
f"{len(f)} finding(s)")
ev("CB-EV-0100-self.md",
"CB-WP-0019 T04.\n| CB-WP-0019 | $9.99 provisional |\n")
check("own-cost accepts a figure marked provisional",
not check_own_cost_not_quoted(tmp))
ev("CB-EV-0100-self.md", "CB-WP-0019 T04.\n| CB-WP-0018 | $28.08 |\n")
check("own-cost accepts quoting an EARLIER pass",
not check_own_cost_not_quoted(tmp))
# and it binds forward: the six passes that PROVE the rule are the
# evidence for it, and must not be edited to satisfy it.
ev("CB-EV-0100-self.md", "CB-WP-0009 T04.\n| CB-WP-0009 | $6.73 |\n")
check("own-cost binds forward, not over the record it rests on",
not check_own_cost_not_quoted(tmp))
os.remove(os.path.join(tmp, "evidence", "CB-EV-0100-self.md"))
# lifecycle: `ready` with work started trips; `active` does not.
def wp(status, tasks):
body = f"---\nid: CB-WP-0100\nstatus: {status}\n---\n"
for t in tasks:
body += f"\n```task\nid: CB-WP-0100-T\nstatus: {t}\npriority: high\n```\n"
with open(os.path.join(tmp, "workplans", "CB-WP-0100-x.md"), "w") as fh:
fh.write(body)
wp("ready", ["done", "todo"])
f = check_workplan_lifecycle(tmp)
check("lifecycle detects `ready` after work has started",
len(f) == 1 and "active" in f[0].detail, f"{len(f)} finding(s)")
wp("active", ["done", "todo"])
check("lifecycle accepts `active` mid-flight",
not check_workplan_lifecycle(tmp))
wp("active", ["done", "cancel"])
f = check_workplan_lifecycle(tmp)
check("lifecycle detects `active` when everything is closed",
len(f) == 1 and "done" in f[0].detail, f"{len(f)} finding(s)")
wp("done", ["done", "todo"])
check("lifecycle detects `done` with an open task",
len(check_workplan_lifecycle(tmp)) == 1)
os.remove(os.path.join(tmp, "workplans", "CB-WP-0100-x.md"))
# gates: an unclassified `all:` dependency trips, and so does an
# entry naming a target the Makefile lacks.
with open(os.path.join(tmp, "Makefile"), "w") as fh:
fh.write("all: coverage newthing\ncoverage:\n\techo\n")
with open(os.path.join(tmp, "gates.toml"), "w") as fh:
fh.write('not_control_gates = []\n\n[[gate]]\nid = "G"\n'
'name = "n"\ntarget = "coverage"\nchecks = "c"\n'
'added = "2026-01-01"\nreview_by = "2026-02-01"\n'
'retire_if = "r"\n')
f = check_gate_registry(tmp)
check("gate registry detects an unclassified all: dependency",
len(f) == 1 and "newthing" in f[0].detail, f"{len(f)} finding(s)")
with open(os.path.join(tmp, "gates.toml"), "w") as fh:
fh.write('not_control_gates = ["newthing", "coverage"]\n\n[[gate]]\nid = "G"\n'
'name = "n"\ntarget = "ghost"\nchecks = "c"\n'
'added = "2026-01-01"\nreview_by = "2026-02-01"\n'
'retire_if = "r"\n')
f = check_gate_registry(tmp)
check("gate registry detects an entry naming a missing target",
len(f) == 1 and "ghost" in f[0].detail, f"{len(f)} finding(s)")
os.unlink(os.path.join(tmp, "gates.toml"))
os.unlink(os.path.join(tmp, "Makefile"))
# evidence: a table verdict trips; the word in prose does not.
with open(os.path.join(tmp, "evidence", "E.md"), "w") as fh:
fh.write("| AC-1 | x | unmeasured |\n"
"the word unmeasured appearing in prose is fine\n")
f = check_evidence_no_unmeasured(tmp)
check("evidence-unmeasured detects a table verdict, not prose",
len(f) == 1, f"{len(f)} finding(s), expected exactly 1")
# tier/chaos: missing tier trips; tier without chaos trips.
with open(os.path.join(tmp, "research", "A.md"), "w") as fh:
fh.write("# survey\nno tier here\n")
with open(os.path.join(tmp, "research", "B.md"), "w") as fh:
fh.write("tier: L (structural L)\n")
f = check_survey_tier_and_chaos(tmp)
rules = sorted(x.rule for x in f)
check("tier/chaos detects both omissions",
rules == ["chaos-recorded", "tier-declared"], f"{rules}")
# self-test: a tool without the flag trips.
with open(os.path.join(tmp, "tools", "silent.py"), "w") as fh:
fh.write("print(42)\n")
f = check_reporting_tools_self_test(tmp)
check("self-test detects a tool lacking --self-test",
len(f) == 1 and "silent.py" in f[0].path, f"{len(f)} finding(s)")
# review trail: approved tier-L survey with no history trips.
with open(os.path.join(tmp, "research", "C.md"), "w") as fh:
fh.write("tier: L (structural L, chaos 3)\nstatus: approved\n")
f = check_review_trail(tmp)
check("review-trail detects a missing challenge/response",
len(f) == 2, f"{len(f)} finding(s), expected 2")
finally:
shutil.rmtree(tmp, ignore_errors=True)
print("loop-lint self-test (positive control)")
ok = True
for name, passed, detail in results:
print(f" [{'ok ' if passed else 'FAIL'}] {name}"
+ (f"{detail}" if detail else ""))
ok &= passed
return 0 if ok else 1
def main():
if "--self-test" in sys.argv:
return self_test()
findings = []
for c in CHECKS:
findings.extend(c())
print("loop-lint — executable InnerLoop rules")
if not findings:
print(" no findings")
return 0
by_rule = {}
for f in findings:
by_rule.setdefault(f.rule, []).append(f)
for rule, fs in sorted(by_rule.items()):
print(f"\n{rule} ({len(fs)}):")
for f in fs:
print(str(f))
print(f"\n{len(findings)} finding(s)")
return 1
if __name__ == "__main__":
sys.exit(main())