Some checks failed
ci / check (push) Has been cancelled
T03: InnerLoop v1.7 plus loop-lint's own-cost check. Six passes under-reported themselves by 30-45%, never once high, and the rule lived only in evidence files having been re-derived three times. The READING is load bearing, not the boundary: CB-WP-0018 T04 applied 're-run the instrument at the moment of quoting' alone and its figure was correct. So the operative instruction is re-run when you quote, and loop-lint fails an evidence file naming its own workplan beside a dollar amount without marking it provisional. It binds forward from this pass. The check fires on seven historical files which ARE the evidence for the rule; making them comply would edit the record to remove the thing it proves -- the same category error as a live fact: tag on a dated measurement, which this pass also hit. Lifecycle, at the maintainer's instruction: ready -> active -> done, where ready means declared and not started. loop-lint fails a workplan that has started and still says ready, one that is active with everything closed, and one that is done with an open task. The first version of that check was WRONG and its own self-test caught it: it stripped the leading status: assuming frontmatter, which silently dropped a real task once the frontmatter said ready or active. Both new checks then fired on this pass's own artifacts and both were right to. T04: CB-EV-0017. The new meta budget's first reading is a breach it caused -- 27% against the 20% line, because this pass cost $31.18 against product passes averaging ~$21. Reported rather than exempted: ADR-0006 D2 covers the instrument repairs but not the rule-writing, and the honest reading is that this should have been two passes. CB-WP-0018 settled at $36.53/95 against $28.08/82 last reported, 30% higher. Seven for seven. make all exits 0. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
483 lines
19 KiB
Python
483 lines
19 KiB
Python
#!/usr/bin/env python3
|
|
"""Executable checks for specs/InnerLoop.md rules (CB-WP-0003 T01).
|
|
|
|
The loop's own rules were prose. A rule nobody can run is a suggestion,
|
|
and the audit in history/260731-inner-loop-rule-audit.md found several
|
|
that were already being violated with no signal. This makes the
|
|
mechanically-checkable ones fail a command.
|
|
|
|
Each check names the InnerLoop rule it enforces. Checks that cannot be
|
|
made mechanical are recorded in the audit as `checkable` or `decorative`
|
|
and are deliberately absent here — see the audit for why.
|
|
|
|
Positive control (InnerLoop v1.1 §Step 5): --self-test asserts each check
|
|
actually detects its failure, using fixtures with known answers. A linter
|
|
that passes everything because its matcher is broken is the same defect
|
|
class as a benchmark timing rejected work.
|
|
|
|
Usage:
|
|
python3 tools/loop-lint.py # lint the repo
|
|
python3 tools/loop-lint.py --self-test # positive control
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
from repo import ROOT as REPO # noqa: E402 (single source of fact, T01)
|
|
LOADABILITY_LIMIT = 400
|
|
|
|
# Artifact classes the loop produces. history/ is an append-only trail
|
|
# (verbatim challenge text is not something to split), so it is exempt.
|
|
LOOP_DIRS = ("specs", "research", "decisions", "evidence", "workplans")
|
|
|
|
|
|
class Finding:
|
|
def __init__(self, rule, path, detail):
|
|
self.rule, self.path, self.detail = rule, path, detail
|
|
|
|
def __str__(self):
|
|
return f" [{self.rule}] {self.path}\n {self.detail}"
|
|
|
|
|
|
def _md_files(root=REPO):
|
|
"""Loop artifacts only.
|
|
|
|
Vendored third-party trees (a baseline harness ships its own
|
|
node_modules) are not artifacts this loop produces, and linting them
|
|
buries the two real findings under eighteen irrelevant ones.
|
|
"""
|
|
for d in LOOP_DIRS:
|
|
base = os.path.join(root, d)
|
|
for dirpath, dirnames, files in os.walk(base):
|
|
dirnames[:] = [x for x in dirnames if x != "node_modules"]
|
|
for f in sorted(files):
|
|
if f.endswith(".md"):
|
|
yield os.path.relpath(os.path.join(dirpath, f), root)
|
|
|
|
|
|
def check_loadability(root=REPO):
|
|
"""§Agentic-efficiency 1 — every loop artifact stays under ~400 lines."""
|
|
out = []
|
|
for rel in _md_files(root):
|
|
with open(os.path.join(root, rel)) as fh:
|
|
n = sum(1 for _ in fh)
|
|
if n > LOADABILITY_LIMIT:
|
|
out.append(
|
|
Finding(
|
|
"loadability",
|
|
rel,
|
|
f"{n} lines exceeds the ~{LOADABILITY_LIMIT}-line limit; "
|
|
f"split and link with relative paths",
|
|
)
|
|
)
|
|
return out
|
|
|
|
|
|
def check_evidence_no_unmeasured(root=REPO):
|
|
"""§Rubric — `unmeasured` is legal in a survey, illegal in an evidence file."""
|
|
out = []
|
|
base = os.path.join(root, "evidence")
|
|
if not os.path.isdir(base):
|
|
return out
|
|
for f in sorted(os.listdir(base)):
|
|
if not f.endswith(".md"):
|
|
continue
|
|
rel = os.path.join("evidence", f)
|
|
for i, line in enumerate(open(os.path.join(root, rel)), 1):
|
|
# A row asserting the verdict, not prose discussing the word.
|
|
if re.search(r"\|\s*unmeasured\s*\|", line):
|
|
out.append(
|
|
Finding("evidence-unmeasured", f"{rel}:{i}",
|
|
"verdict `unmeasured` in an evidence table")
|
|
)
|
|
return out
|
|
|
|
|
|
def check_own_cost_not_quoted(root=REPO):
|
|
"""§Quoting a cost — an evidence file may not quote its own pass's
|
|
cost as final.
|
|
|
|
Six passes under-reported themselves by 30-45%, never once high, so a
|
|
self-quoted figure is not a rounding error but a known bias. The check
|
|
is deliberately narrow: it fires only when a line names the file's OWN
|
|
workplan beside a dollar amount and does not mark it provisional.
|
|
Quoting an earlier pass is exactly what the rule asks for.
|
|
|
|
**Binds forward, from the pass that wrote it down.** The rule was
|
|
written in CB-WP-0019 after six passes had each under-reported
|
|
themselves, and those six evidence files are the *evidence for the
|
|
rule*. Firing on them would demand the record be edited to remove the
|
|
thing it proves — the same category error as putting a live `fact:`
|
|
tag on a dated measurement. So the check applies from CB-WP-0019 on,
|
|
and the older figures stay as they were reported.
|
|
"""
|
|
BINDS_FROM = 19
|
|
out = []
|
|
base = os.path.join(root, "evidence")
|
|
if not os.path.isdir(base):
|
|
return out
|
|
for f in sorted(os.listdir(base)):
|
|
if not f.endswith(".md"):
|
|
continue
|
|
rel = os.path.join("evidence", f)
|
|
text = open(os.path.join(root, rel)).read()
|
|
# The pass an evidence file belongs to is named in its header.
|
|
m = re.search(r"\b(CB-WP-\d{4})\b", text)
|
|
if not m:
|
|
continue
|
|
own = m.group(1)
|
|
if int(own.rsplit("-", 1)[1]) < BINDS_FROM:
|
|
continue
|
|
for i, line in enumerate(text.splitlines(), 1):
|
|
if own not in line or "$" not in line:
|
|
continue
|
|
if "provisional" in line.lower() or "not quoted" in line.lower():
|
|
continue
|
|
out.append(
|
|
Finding("own-cost", f"{rel}:{i}",
|
|
f"quotes {own}'s own cost as final; re-run the "
|
|
f"instrument and quote a settled pass, or mark it "
|
|
f"provisional")
|
|
)
|
|
return out
|
|
|
|
|
|
def check_workplan_lifecycle(root=REPO):
|
|
"""§Workplan lifecycle — `ready` means declared and NOT started.
|
|
|
|
A workplan with work done in it that still says `ready` answers the
|
|
wrong question: the status should say whether anyone is on it, not
|
|
only whether it is finished.
|
|
"""
|
|
out = []
|
|
base = os.path.join(root, "workplans")
|
|
if not os.path.isdir(base):
|
|
return out
|
|
for f in sorted(os.listdir(base)):
|
|
if not f.endswith(".md"):
|
|
continue
|
|
rel = os.path.join("workplans", f)
|
|
text = open(os.path.join(root, rel)).read()
|
|
m = re.search(r"^status:\s*(\S+)", text, re.M)
|
|
if not m:
|
|
continue
|
|
status = m.group(1)
|
|
# Parse the ```task blocks, not every `status:` in the file. The
|
|
# first version stripped the leading match assuming it was the
|
|
# frontmatter — which silently dropped a real task the moment the
|
|
# frontmatter said `ready` or `active`, because those do not match
|
|
# the task vocabulary. Caught by this check's own self-test.
|
|
tasks = [
|
|
t.group(1)
|
|
for block in re.findall(r"```task\n(.*?)```", text, re.S)
|
|
for t in [re.search(r"^status:\s*(\S+)", block, re.M)]
|
|
if t
|
|
]
|
|
if not tasks:
|
|
continue
|
|
closed = sum(1 for t in tasks if t in ("done", "cancel"))
|
|
started = closed > 0
|
|
if status == "ready" and started:
|
|
out.append(Finding("lifecycle", rel,
|
|
"still `ready` but work has started — use `active`"))
|
|
elif status == "active" and closed == len(tasks):
|
|
out.append(Finding("lifecycle", rel,
|
|
"every task is closed but status is `active` — use `done`"))
|
|
elif status == "done" and closed != len(tasks):
|
|
out.append(Finding("lifecycle", rel,
|
|
f"`done` with {len(tasks) - closed} task(s) still open"))
|
|
return out
|
|
|
|
|
|
def check_survey_tier_and_chaos(root=REPO):
|
|
"""§Loop tiers — tier declared, and the chaos roll recorded every time."""
|
|
out = []
|
|
base = os.path.join(root, "research")
|
|
if not os.path.isdir(base):
|
|
return out
|
|
for f in sorted(os.listdir(base)):
|
|
if not f.endswith(".md"):
|
|
continue
|
|
rel = os.path.join("research", f)
|
|
text = open(os.path.join(root, rel)).read()
|
|
if not re.search(r"^tier:\s*[SML]\b", text, re.M):
|
|
out.append(Finding("tier-declared", rel, "no `tier:` declaration"))
|
|
elif "chaos" not in text.lower():
|
|
out.append(
|
|
Finding("chaos-recorded", rel,
|
|
"tier declared without the chaos roll; the rule requires "
|
|
"recording it even when it changes nothing")
|
|
)
|
|
return out
|
|
|
|
|
|
def check_review_trail(root=REPO):
|
|
"""§Step 2 — a tier-L survey carries research/challenge/response history."""
|
|
out = []
|
|
base = os.path.join(root, "research")
|
|
hist = os.path.join(root, "history")
|
|
if not (os.path.isdir(base) and os.path.isdir(hist)):
|
|
return out
|
|
files = os.listdir(hist)
|
|
for f in sorted(os.listdir(base)):
|
|
if not f.endswith(".md"):
|
|
continue
|
|
rel = os.path.join("research", f)
|
|
text = open(os.path.join(root, rel)).read()
|
|
if not re.search(r"^tier:\s*L\b", text, re.M):
|
|
continue
|
|
if not re.search(r"^status:\s*approved", text, re.M):
|
|
continue
|
|
for kind in ("challenge", "response"):
|
|
if not any(x.endswith(f"-{kind}.md") and kind in x for x in files):
|
|
out.append(
|
|
Finding("review-trail", rel,
|
|
f"tier-L approved survey with no history/*-{kind}.md")
|
|
)
|
|
return out
|
|
|
|
|
|
def check_reporting_tools_self_test(root=REPO):
|
|
"""§Step 5 v1.1 — every tool that reports a number exposes --self-test."""
|
|
out = []
|
|
base = os.path.join(root, "tools")
|
|
if not os.path.isdir(base):
|
|
return out
|
|
for f in sorted(os.listdir(base)):
|
|
if not f.endswith(".py"):
|
|
continue
|
|
rel = os.path.join("tools", f)
|
|
text = open(os.path.join(root, rel)).read()
|
|
if "--self-test" not in text:
|
|
out.append(
|
|
Finding("self-test", rel,
|
|
"reporting tool with no --self-test entry point; "
|
|
"nothing verifies its positive control still works")
|
|
)
|
|
return out
|
|
|
|
|
|
def check_gate_registry(root=REPO):
|
|
"""ADR-0006 D3 — every control gate is in `gates.toml`, and every
|
|
entry names a real target.
|
|
|
|
The failure this prevents is drift in the direction nobody notices: a
|
|
gate added to `make all` with no registry entry never acquires a
|
|
review date, which is how five mechanisms accumulated with no way to
|
|
retire any of them.
|
|
"""
|
|
out = []
|
|
registry = os.path.join(root, "gates.toml")
|
|
makefile = os.path.join(root, "Makefile")
|
|
if not (os.path.exists(registry) and os.path.exists(makefile)):
|
|
return out
|
|
try:
|
|
import tomllib
|
|
except ModuleNotFoundError: # pragma: no cover
|
|
return out
|
|
|
|
with open(registry, "rb") as fh:
|
|
data = tomllib.load(fh)
|
|
gates = data.get("gate") or []
|
|
if not gates:
|
|
return [Finding("gates", "gates.toml", "registry contains no gates")]
|
|
registered = {g.get("target") for g in gates if g.get("target")}
|
|
exempt = set(data.get("not_control_gates") or [])
|
|
|
|
text = open(makefile).read()
|
|
m = re.search(r"^all:(.*)$", text, re.M)
|
|
deps = m.group(1).split() if m else []
|
|
targets = {ln.split(":", 1)[0].strip() for ln in text.splitlines()
|
|
if ln and not ln[0].isspace() and ":" in ln and not ln.startswith(".")}
|
|
|
|
for dep in deps:
|
|
if dep not in registered and dep not in exempt:
|
|
out.append(Finding(
|
|
"gates", "gates.toml",
|
|
f"`make all` runs {dep!r}, which is neither a registered "
|
|
f"control gate nor listed in not_control_gates — classify it, "
|
|
f"so it cannot acquire permanence without a review date"))
|
|
for target in sorted(registered):
|
|
if target not in targets:
|
|
out.append(Finding(
|
|
"gates", "gates.toml",
|
|
f"entry names target {target!r}, which the Makefile lacks"))
|
|
return out
|
|
|
|
|
|
CHECKS = (
|
|
check_loadability,
|
|
check_evidence_no_unmeasured,
|
|
check_own_cost_not_quoted,
|
|
check_workplan_lifecycle,
|
|
check_survey_tier_and_chaos,
|
|
check_review_trail,
|
|
check_reporting_tools_self_test,
|
|
check_gate_registry,
|
|
)
|
|
|
|
|
|
def self_test():
|
|
"""Each check must DETECT its failure, not merely run."""
|
|
import shutil
|
|
import tempfile
|
|
|
|
results = []
|
|
|
|
def check(name, ok, detail=""):
|
|
results.append((name, ok, detail))
|
|
|
|
tmp = tempfile.mkdtemp()
|
|
try:
|
|
for d in LOOP_DIRS + ("tools", "history"):
|
|
os.makedirs(os.path.join(tmp, d), exist_ok=True)
|
|
|
|
# loadability: 401 lines must trip, 400 must not.
|
|
with open(os.path.join(tmp, "specs", "Big.md"), "w") as fh:
|
|
fh.write("x\n" * (LOADABILITY_LIMIT + 1))
|
|
with open(os.path.join(tmp, "specs", "Ok.md"), "w") as fh:
|
|
fh.write("x\n" * LOADABILITY_LIMIT)
|
|
f = check_loadability(tmp)
|
|
check("loadability detects overlong artifact",
|
|
len(f) == 1 and "Big.md" in f[0].path,
|
|
f"{len(f)} finding(s)")
|
|
|
|
# own-cost: a file quoting its OWN pass beside a dollar amount
|
|
# trips; the same line marked provisional does not; and a file
|
|
# quoting an EARLIER pass does not, because that is the rule.
|
|
def ev(name, body):
|
|
with open(os.path.join(tmp, "evidence", name), "w") as fh:
|
|
fh.write(body)
|
|
ev("CB-EV-0100-self.md", "CB-WP-0019 T04.\n| CB-WP-0019 | $9.99 |\n")
|
|
f = check_own_cost_not_quoted(tmp)
|
|
check("own-cost detects a pass quoting itself", len(f) == 1,
|
|
f"{len(f)} finding(s)")
|
|
ev("CB-EV-0100-self.md",
|
|
"CB-WP-0019 T04.\n| CB-WP-0019 | $9.99 provisional |\n")
|
|
check("own-cost accepts a figure marked provisional",
|
|
not check_own_cost_not_quoted(tmp))
|
|
ev("CB-EV-0100-self.md", "CB-WP-0019 T04.\n| CB-WP-0018 | $28.08 |\n")
|
|
check("own-cost accepts quoting an EARLIER pass",
|
|
not check_own_cost_not_quoted(tmp))
|
|
# and it binds forward: the six passes that PROVE the rule are the
|
|
# evidence for it, and must not be edited to satisfy it.
|
|
ev("CB-EV-0100-self.md", "CB-WP-0009 T04.\n| CB-WP-0009 | $6.73 |\n")
|
|
check("own-cost binds forward, not over the record it rests on",
|
|
not check_own_cost_not_quoted(tmp))
|
|
os.remove(os.path.join(tmp, "evidence", "CB-EV-0100-self.md"))
|
|
|
|
# lifecycle: `ready` with work started trips; `active` does not.
|
|
def wp(status, tasks):
|
|
body = f"---\nid: CB-WP-0100\nstatus: {status}\n---\n"
|
|
for t in tasks:
|
|
body += f"\n```task\nid: CB-WP-0100-T\nstatus: {t}\npriority: high\n```\n"
|
|
with open(os.path.join(tmp, "workplans", "CB-WP-0100-x.md"), "w") as fh:
|
|
fh.write(body)
|
|
wp("ready", ["done", "todo"])
|
|
f = check_workplan_lifecycle(tmp)
|
|
check("lifecycle detects `ready` after work has started",
|
|
len(f) == 1 and "active" in f[0].detail, f"{len(f)} finding(s)")
|
|
wp("active", ["done", "todo"])
|
|
check("lifecycle accepts `active` mid-flight",
|
|
not check_workplan_lifecycle(tmp))
|
|
wp("active", ["done", "cancel"])
|
|
f = check_workplan_lifecycle(tmp)
|
|
check("lifecycle detects `active` when everything is closed",
|
|
len(f) == 1 and "done" in f[0].detail, f"{len(f)} finding(s)")
|
|
wp("done", ["done", "todo"])
|
|
check("lifecycle detects `done` with an open task",
|
|
len(check_workplan_lifecycle(tmp)) == 1)
|
|
os.remove(os.path.join(tmp, "workplans", "CB-WP-0100-x.md"))
|
|
|
|
# gates: an unclassified `all:` dependency trips, and so does an
|
|
# entry naming a target the Makefile lacks.
|
|
with open(os.path.join(tmp, "Makefile"), "w") as fh:
|
|
fh.write("all: coverage newthing\ncoverage:\n\techo\n")
|
|
with open(os.path.join(tmp, "gates.toml"), "w") as fh:
|
|
fh.write('not_control_gates = []\n\n[[gate]]\nid = "G"\n'
|
|
'name = "n"\ntarget = "coverage"\nchecks = "c"\n'
|
|
'added = "2026-01-01"\nreview_by = "2026-02-01"\n'
|
|
'retire_if = "r"\n')
|
|
f = check_gate_registry(tmp)
|
|
check("gate registry detects an unclassified all: dependency",
|
|
len(f) == 1 and "newthing" in f[0].detail, f"{len(f)} finding(s)")
|
|
with open(os.path.join(tmp, "gates.toml"), "w") as fh:
|
|
fh.write('not_control_gates = ["newthing", "coverage"]\n\n[[gate]]\nid = "G"\n'
|
|
'name = "n"\ntarget = "ghost"\nchecks = "c"\n'
|
|
'added = "2026-01-01"\nreview_by = "2026-02-01"\n'
|
|
'retire_if = "r"\n')
|
|
f = check_gate_registry(tmp)
|
|
check("gate registry detects an entry naming a missing target",
|
|
len(f) == 1 and "ghost" in f[0].detail, f"{len(f)} finding(s)")
|
|
os.unlink(os.path.join(tmp, "gates.toml"))
|
|
os.unlink(os.path.join(tmp, "Makefile"))
|
|
|
|
# evidence: a table verdict trips; the word in prose does not.
|
|
with open(os.path.join(tmp, "evidence", "E.md"), "w") as fh:
|
|
fh.write("| AC-1 | x | unmeasured |\n"
|
|
"the word unmeasured appearing in prose is fine\n")
|
|
f = check_evidence_no_unmeasured(tmp)
|
|
check("evidence-unmeasured detects a table verdict, not prose",
|
|
len(f) == 1, f"{len(f)} finding(s), expected exactly 1")
|
|
|
|
# tier/chaos: missing tier trips; tier without chaos trips.
|
|
with open(os.path.join(tmp, "research", "A.md"), "w") as fh:
|
|
fh.write("# survey\nno tier here\n")
|
|
with open(os.path.join(tmp, "research", "B.md"), "w") as fh:
|
|
fh.write("tier: L (structural L)\n")
|
|
f = check_survey_tier_and_chaos(tmp)
|
|
rules = sorted(x.rule for x in f)
|
|
check("tier/chaos detects both omissions",
|
|
rules == ["chaos-recorded", "tier-declared"], f"{rules}")
|
|
|
|
# self-test: a tool without the flag trips.
|
|
with open(os.path.join(tmp, "tools", "silent.py"), "w") as fh:
|
|
fh.write("print(42)\n")
|
|
f = check_reporting_tools_self_test(tmp)
|
|
check("self-test detects a tool lacking --self-test",
|
|
len(f) == 1 and "silent.py" in f[0].path, f"{len(f)} finding(s)")
|
|
|
|
# review trail: approved tier-L survey with no history trips.
|
|
with open(os.path.join(tmp, "research", "C.md"), "w") as fh:
|
|
fh.write("tier: L (structural L, chaos 3)\nstatus: approved\n")
|
|
f = check_review_trail(tmp)
|
|
check("review-trail detects a missing challenge/response",
|
|
len(f) == 2, f"{len(f)} finding(s), expected 2")
|
|
finally:
|
|
shutil.rmtree(tmp, ignore_errors=True)
|
|
|
|
print("loop-lint self-test (positive control)")
|
|
ok = True
|
|
for name, passed, detail in results:
|
|
print(f" [{'ok ' if passed else 'FAIL'}] {name}"
|
|
+ (f" — {detail}" if detail else ""))
|
|
ok &= passed
|
|
return 0 if ok else 1
|
|
|
|
|
|
def main():
|
|
if "--self-test" in sys.argv:
|
|
return self_test()
|
|
|
|
findings = []
|
|
for c in CHECKS:
|
|
findings.extend(c())
|
|
|
|
print("loop-lint — executable InnerLoop rules")
|
|
if not findings:
|
|
print(" no findings")
|
|
return 0
|
|
by_rule = {}
|
|
for f in findings:
|
|
by_rule.setdefault(f.rule, []).append(f)
|
|
for rule, fs in sorted(by_rule.items()):
|
|
print(f"\n{rule} ({len(fs)}):")
|
|
for f in fs:
|
|
print(str(f))
|
|
print(f"\n{len(findings)} finding(s)")
|
|
return 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|