AM-4c is withdrawn from the acceptance table and retained as a reported diagnostic. GameKernel §5a carries the argument. The ratio has no monotone better direction. INTENT's rule is "own the semantics, assimilate the implementation": rising can mean owning semantics properly or reimplementing what should have been assimilated; falling can mean leverage or dependency bloat. A target requires knowing which way is better. It is also redundant — AM-4a/AM-4b bound the denominator and AM-2 bounds own-source density, so AM-4c is a ratio of two already-targeted quantities. Measured at withdrawal: 1,426 own lines per 100k third-party (shipped), 1,107 (dev). make dep-weight now prints both, labelled diagnostic — the row was never actually reported before. M-D1-MUT keeps AM-4c in its denominator on purpose and says so in the output. Dropping it would move the score 7/14 -> 7/13 without enforcing anything: a score improved by deleting the question. Decided before Phase B deliberately, since ADR-0005 predicts own-source growth that will move this ratio; deciding after would be the retarget §Step 4 forbids. A T01 correction found here. The AM-6 gate failed inside `make all` at 38,753 ev/s against 341,280 in isolation — a 9x drop, because cargo test runs binaries and threads concurrently. A throughput assertion inside a parallel harness measures contention, not throughput. T01's measurement was valid; its gate placement was not. Fixed by running it only where valid — #[ignore] plus `make am6` in release with --test-threads=1, now 2.0M ev/s at 20.2x headroom — and not by lowering the target, which T01 forbade. My first attempt did drift that way, adding a debug "sanity floor" of 50,000, and was backed out: a second threshold is still a second chance to tune. The mutation then went SURVIVED on the first run after the move. 4,000 black_box iterations were calibrated against debug's 3.4x headroom and are invisible against release's 20x. Raised to 100,000; back to red. A weak mutation is not a fixed property of a row — it can become weak when the row's measurement conditions change. Tier S (amends one row, creates no capability), chaos d4=2, no override. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
223 lines
8 KiB
Python
Executable file
223 lines
8 KiB
Python
Executable file
#!/usr/bin/env python3
|
|
"""AM-4: third-party dependency weight, measured as source under audit.
|
|
|
|
Crate count is a poor cross-ecosystem proxy — Rust splits crates far more
|
|
finely than npm, so "33 crates vs 120 npm packages" flatters us in one
|
|
direction and a low crate-count target punishes us in the other. What the
|
|
count stands in for is how much third-party source a reviewer would have
|
|
to audit. This measures that directly, in two configurations:
|
|
|
|
shipped-runtime cargo build --no-default-features (what a game ships)
|
|
dev-toolchain cargo build (adds scenario YAML)
|
|
|
|
Positive control (InnerLoop v1.0 §Step 5): every crate in the dependency
|
|
graph must be located on disk and produce a non-zero line count. A crate
|
|
that cannot be found is reported and the run exits non-zero rather than
|
|
silently under-reporting the total — under-reporting is the exact
|
|
direction this metric could be gamed.
|
|
|
|
Usage: python3 tools/dep-weight.py [--json] [--self-test]
|
|
"""
|
|
|
|
import glob
|
|
import json
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
|
|
from repo import cargo_bin, enter_root
|
|
|
|
PACKAGE = "games-ground"
|
|
CONFIGS = {
|
|
"shipped-runtime": ["--no-default-features"],
|
|
"dev-toolchain": [],
|
|
}
|
|
|
|
# AM-4a / AM-4b targets from specs/GameKernel.md §4. Breaching one fails
|
|
# the build: a gate that only reports is a suggestion.
|
|
TARGETS = {
|
|
"shipped-runtime": 250_000,
|
|
"dev-toolchain": 350_000,
|
|
}
|
|
|
|
|
|
def crates(extra_args):
|
|
"""Third-party crates in the normal (non-dev) dependency graph."""
|
|
# T01: locate cargo rather than demanding the caller export PATH. The
|
|
# error below is kept as a positive control — it should now be
|
|
# unreachable on a machine with rustup installed, and a control that
|
|
# never fires is still cheaper than a regression.
|
|
cargo = cargo_bin()
|
|
if not cargo:
|
|
print(
|
|
"ERROR: cargo not found on PATH or in ~/.cargo/bin — is rustup installed?",
|
|
file=sys.stderr,
|
|
)
|
|
sys.exit(1)
|
|
out = subprocess.run(
|
|
[cargo, "tree", "-p", PACKAGE, "--edges", "normal", "--prefix", "none"]
|
|
+ extra_args,
|
|
capture_output=True,
|
|
text=True,
|
|
check=True,
|
|
).stdout
|
|
found = {}
|
|
for line in out.splitlines():
|
|
parts = line.split()
|
|
if len(parts) < 2 or not parts[1].startswith("v"):
|
|
continue
|
|
name, version = parts[0], parts[1].lstrip("v")
|
|
# Path dependencies are our own code, not third-party.
|
|
if "(/" in line:
|
|
continue
|
|
found[name] = version
|
|
return found
|
|
|
|
|
|
def source_lines(name, version):
|
|
"""Lines of Rust in the vendored source for one crate."""
|
|
roots = glob.glob(os.path.expanduser("~/.cargo/registry/src/*/"))
|
|
for root in roots:
|
|
# Version may carry a build suffix (e.g. 0.9.34+deprecated).
|
|
for d in glob.glob(f"{root}{name}-{version}*/") + glob.glob(f"{root}{name}-*/"):
|
|
total = 0
|
|
for dirpath, _, files in os.walk(d):
|
|
for f in files:
|
|
if f.endswith(".rs"):
|
|
try:
|
|
with open(os.path.join(dirpath, f), "rb") as fh:
|
|
total += fh.read().count(b"\n")
|
|
except OSError:
|
|
pass
|
|
if total:
|
|
return total
|
|
return 0
|
|
|
|
|
|
def self_test():
|
|
"""Each assertion pins a failure this tool must detect.
|
|
|
|
The controls that matter here are: an unlocatable crate must not be
|
|
silently counted as zero lines (that under-reports, the direction this
|
|
metric could be gamed), and a target breach must fail rather than
|
|
merely print.
|
|
"""
|
|
results = []
|
|
|
|
def check(name, ok, detail=""):
|
|
results.append((name, ok, detail))
|
|
|
|
# A crate that does not exist must measure zero, so the caller's
|
|
# `lines == 0` guard fires rather than silently shrinking the total.
|
|
check("unlocatable crate measures zero (so the guard fires)",
|
|
source_lines("definitely-not-a-real-crate-xyz", "9.9.9") == 0)
|
|
|
|
# A crate we do depend on must measure non-zero, or the guard above
|
|
# would fire on everything and the tool would never report at all.
|
|
real = source_lines("serde", "1")
|
|
check("a real vendored crate measures non-zero", real > 0,
|
|
f"{real:,} lines")
|
|
|
|
# Targets must be present and numeric — a missing target would make
|
|
# the breach check vacuous.
|
|
# T01: this tool is useless without cargo, and used to demand the caller
|
|
# put it on PATH. Assert it resolves unaided.
|
|
check("cargo resolves without caller PATH setup", bool(cargo_bin()),
|
|
cargo_bin() or "NOT FOUND")
|
|
|
|
check("targets defined for every configuration",
|
|
set(TARGETS) == set(CONFIGS) and all(
|
|
isinstance(v, int) and v > 0 for v in TARGETS.values()),
|
|
f"{TARGETS}")
|
|
|
|
print("dep-weight self-test (positive control)")
|
|
ok = True
|
|
for name, passed, detail in results:
|
|
print(f" [{'ok ' if passed else 'FAIL'}] {name}"
|
|
+ (f" — {detail}" if detail else ""))
|
|
ok &= passed
|
|
return 0 if ok else 1
|
|
|
|
|
|
def main():
|
|
# T01: `cargo tree` and the own-source walk are both repo-relative.
|
|
enter_root()
|
|
if "--self-test" in sys.argv:
|
|
return self_test()
|
|
|
|
report = {}
|
|
missing = []
|
|
for label, args in CONFIGS.items():
|
|
found = crates(args)
|
|
per_crate = {}
|
|
for name, version in sorted(found.items()):
|
|
lines = source_lines(name, version)
|
|
if lines == 0:
|
|
missing.append(f"{name} {version} ({label})")
|
|
per_crate[name] = lines
|
|
report[label] = {
|
|
"crates": len(found),
|
|
"third_party_loc": sum(per_crate.values()),
|
|
"per_crate": per_crate,
|
|
}
|
|
|
|
own = 0
|
|
for base in ("crates", "games", "tools"):
|
|
for dirpath, _, files in os.walk(base):
|
|
if "target" in dirpath.split(os.sep):
|
|
continue
|
|
for f in files:
|
|
if f.endswith(".rs"):
|
|
with open(os.path.join(dirpath, f), "rb") as fh:
|
|
own += fh.read().count(b"\n")
|
|
report["own_loc"] = own
|
|
|
|
if "--json" in sys.argv:
|
|
print(json.dumps(report, indent=2))
|
|
else:
|
|
print("AM-4 dependency weight")
|
|
print(f" own source {own:>9,} lines")
|
|
for label in CONFIGS:
|
|
r = report[label]
|
|
limit = TARGETS[label]
|
|
mark = "ok " if r["third_party_loc"] <= limit else "FAIL"
|
|
print(
|
|
f" {label:<18}{r['crates']:>3} crates "
|
|
f"{r['third_party_loc']:>9,} lines third-party "
|
|
f"[{mark} target {limit:,}]"
|
|
)
|
|
# AM-4c: own source per 100k third-party lines. A DIAGNOSTIC, not a
|
|
# target — see specs/GameKernel.md §5 and CB-WP-0006 T04. The ratio
|
|
# has no monotone better direction, so it cannot carry a threshold.
|
|
for label in CONFIGS:
|
|
tp = report[label]["third_party_loc"]
|
|
if tp:
|
|
print(f" AM-4c {label:<18}{own / (tp / 100_000):>9,.0f} own "
|
|
f"lines per 100k third-party (diagnostic, not targeted)")
|
|
delta = (
|
|
report["dev-toolchain"]["third_party_loc"]
|
|
- report["shipped-runtime"]["third_party_loc"]
|
|
)
|
|
print(f" scenario tooling costs {delta:>9,} lines (dev only)")
|
|
|
|
if missing:
|
|
# Positive control: a crate we could not measure would silently
|
|
# shrink the total, so refuse to report rather than under-report.
|
|
print("\nERROR — source not found for:", ", ".join(missing), file=sys.stderr)
|
|
return 1
|
|
|
|
breached = [
|
|
(label, report[label]["third_party_loc"], limit)
|
|
for label, limit in TARGETS.items()
|
|
if report[label]["third_party_loc"] > limit
|
|
]
|
|
for label, actual, limit in breached:
|
|
print(
|
|
f"\nFAIL AM-4 — {label}: {actual:,} lines exceeds target {limit:,}",
|
|
file=sys.stderr,
|
|
)
|
|
return 1 if breached else 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|