# One command surface (InnerLoop §agentic-efficiency #3). Deterministic, # greppable output; precursor of the `cb` CLI. # # CB-WP-0004 T01: every target here runs from a clean shell, from any # directory, with no prefix. Invoke as `make -C ` from # elsewhere. No target requires `cd` or `export PATH` — CB-RES-0003 # measured 84 turns and $15.33 spent on exactly those two prefixes. # Absolute path to this Makefile's directory, so recipes never depend on # the caller's working directory. REPO := $(patsubst %/,%,$(dir $(abspath $(lastword $(MAKEFILE_LIST))))) # Locate cargo instead of requiring it on the inherited PATH. Mirrors # tools/repo.py:cargo_bin() — kept in sync by `make env-test`. CARGO := $(firstword $(shell command -v cargo 2>/dev/null) \ $(wildcard $(HOME)/.cargo/bin/cargo) \ $(wildcard /usr/local/cargo/bin/cargo) \ cargo) export PATH := $(dir $(CARGO)):$(PATH) PY := python3 TOOLS := $(REPO)/tools # Every cargo recipe runs at the repo root; the shell does not persist cd. IN_REPO := cd $(REPO) && .PHONY: check test sim bench bench-test coverage dep-weight cost cost-test cost-pin cost-budget shape-budget cost-mix loop-lint self-tests env-test task-done status facts-check facts-gen mutation-check size-metrics runtime-metrics build-time am6 am7 am8 edition-check replay-test loc play gate-review all ## fmt + clippy (deny warnings) + HashMap deny-lint check: $(IN_REPO) $(CARGO) fmt --all --check $(IN_REPO) $(CARGO) clippy --workspace --all-targets -- -D warnings ## play GROUND from the terminal (INTENT stage 0's CLI player). ## `make play ARGS="--all-bots --bot random"` to watch one instead. play: $(IN_REPO) $(CARGO) run -q -p cb-play -- $(ARGS) ## which control gates are due for a keep-or-kill argument (ADR-0006 D3) gate-review: $(PY) $(TOOLS)/gate-review.py ## unit + scenario-format tests test: $(IN_REPO) $(CARGO) test --workspace ## run all GROUND scenarios through cb-sim dep-weight: $(PY) $(TOOLS)/dep-weight.py coverage: $(PY) $(TOOLS)/rule-coverage.py # K10: a bundle must re-execute, and must be able to fail. Four controls # from ADR-0005 §6, including truncate-by-one-byte and mutated-seed. replay-test: $(PY) $(TOOLS)/replay-test.py # AM-6 gate. Runs in RELEASE, where headroom is ~24x; the same assertion # in debug has ~3.4x and flaked under load. The target is unchanged — only # where it is measured. am6: $(IN_REPO) $(CARGO) test --release -p games-ground --all-features \ am6_throughput -- --ignored --nocapture --test-threads=1 # ADR-0011 D2: is the vendored edition still what ground-game published? # Reports "upstream not checked out" as its own outcome, never a pass. edition-check: $(PY) $(TOOLS)/edition-check.py # AM-8 N=10 determinism gate. One scenario, ten same-seed replays, all # compared to the first. Not all 25: `make sim` already runs K8's double- # run over every scenario, and repeating that eight more times costs 47 s # per build to re-answer a question the second run already answered. The # extra runs exist for the probabilistic class, and one workload gives # that class its ten samples — see scenario::run_n. am8: $(IN_REPO) $(CARGO) run -q -p cb-sim -- --runs 10 \ $(REPO)/scenarios/ground/gr-r06-round-resolve.yaml # AM-7 scaling gate. Same release / single-thread reasoning as am6 and more # so: a *ratio* of two timings taken under varying contention is worse than # one reading, because the noise multiplies rather than cancels. am7: $(IN_REPO) $(CARGO) test --release -p games-ground --all-features \ am7_cost_per_event -- --ignored --nocapture --test-threads=1 # AM-9 peak RSS (fast, gated). AM-5 needs a clean build — see build-time. runtime-metrics: $(PY) $(TOOLS)/runtime-metrics.py --fast # AM-5: clean release build, ~90s, into a throwaway CARGO_TARGET_DIR so the # working cache survives. Reported not gated per GameKernel §5. build-time: $(PY) $(TOOLS)/runtime-metrics.py # AM-2 (M-D1-SPL, AM-1's anti-gaming pair) and AM-3. size-metrics: $(PY) $(TOOLS)/size-metrics.py # M-D2-CST (specs/CostAccounting.md). cost-test is the positive control and # runs first: a cost number from an unverified collector is void. cost: cost-test $(PY) $(TOOLS)/cb-cost.py --composition --by-task cost-test: $(PY) $(TOOLS)/cb-cost.py --self-test # InnerLoop rules that are mechanically checkable (CB-WP-0003 T01). loop-lint: $(PY) $(TOOLS)/loop-lint.py # Positive control for every reporting tool, per InnerLoop v1.1 Step 5. self-tests: $(PY) $(TOOLS)/cb-cost.py --self-test $(PY) $(TOOLS)/loop-lint.py --self-test $(PY) $(TOOLS)/rule-coverage.py --self-test $(PY) $(TOOLS)/dep-weight.py --self-test $(PY) $(TOOLS)/repo.py --self-test $(PY) $(TOOLS)/task-done.py --self-test $(PY) $(TOOLS)/status.py --self-test $(PY) $(TOOLS)/facts.py --self-test $(PY) $(TOOLS)/mutation-check.py --self-test $(PY) $(TOOLS)/size-metrics.py --self-test $(PY) $(TOOLS)/runtime-metrics.py --self-test $(PY) $(TOOLS)/replay-test.py --self-test $(PY) $(TOOLS)/design.py --self-test cargo run --release -q -p games-ground --example difficulty -- --self-test $(PY) $(TOOLS)/edition-check.py --self-test # T01 positive control: prove the environment fix, do not assume it. Runs # every tool from a foreign working directory with a PATH that has no # cargo on it. Before T01 this failed; if it fails again, the friction is # back and CB-EV-0003's measurement is invalid. env-test: @cd / && env PATH=/usr/bin:/bin $(PY) $(TOOLS)/repo.py --self-test @cd / && env PATH=/usr/bin:/bin $(PY) $(TOOLS)/rule-coverage.py --self-test >/dev/null \ && echo " [ok ] rule-coverage runs from / with no cargo on PATH" @cd / && env PATH=/usr/bin:/bin $(PY) $(TOOLS)/dep-weight.py --self-test >/dev/null \ && echo " [ok ] dep-weight runs from / with no cargo on PATH" @cd / && env PATH=/usr/bin:/bin $(PY) $(TOOLS)/cb-cost.py --self-test >/dev/null \ && echo " [ok ] cb-cost runs from / with no cargo on PATH" @cd / && env PATH=/usr/bin:/bin $(PY) $(TOOLS)/loop-lint.py --self-test >/dev/null \ && echo " [ok ] loop-lint runs from / with no cargo on PATH" @$(MAKE) -C $(REPO) coverage >/dev/null \ && echo " [ok ] make -C works from any directory" # M-D1-MUT (CB-WP-0005 T02): invert each acceptance row's property and # require the verifying command to go red. Deliberately NOT in `make all`: # it rebuilds the workspace once per mutated row. Run it on demand and in # CI, not in the inner loop. mutation-check: $(PY) $(TOOLS)/mutation-check.py $(ARGS) # T04: single source of fact (InnerLoop v1.2) — the DFD gate. # facts.toml is GENERATED; facts-check fails if it disagrees with the # instruments, or if a tagged artifact disagrees with it. facts-check: $(PY) $(TOOLS)/facts.py --check facts-gen: $(PY) $(TOOLS)/facts.py --gen # CB-WP-0025 T06: the difficulty table (specs/RetrospectiveAnalysis.md §4). # Winnable fraction from the solver plus a PLURAL policy panel -- a single # policy's win rate may not be reported as a difficulty (§4.1). difficulty: @cargo run --release -q -p games-ground --example difficulty # CB-WP-0022 T05: the design-finding register, reported over # specs/GroundRules.md. Shows the QUEUE by default; the log of closed # findings is a line, not a listing, because a default view that mixes # them loses the queue property (ADR-0012 D5). design: @$(PY) $(TOOLS)/design.py # T03: one-shot orientation — workplans, next task, spend, fast gates. # Cheap by design: no build. Start a session with this instead of grepping. status: @$(PY) $(TOOLS)/status.py # T02: close a task — flip the workplan file, read the *measured* cost # from the transcripts, push the hub event with real numbers. Refuses on an # unknown or already-done task, and refuses to report an estimate. # make task-done T=CB-WP-0004-T02 task-done: @test -n "$(T)" || { echo "usage: make task-done T=CB-WP-0004-T02" >&2; exit 2; } $(PY) $(TOOLS)/task-done.py $(T) $(ARGS) # CB-WP-0007 T03 / InnerLoop v1.5: session shape for the window since the # last commit. Deliberately NOT in `make all` — failing the build on # context would block committing, and committing is the natural point to # compact. A gate that blocks the remedy is a trap. shape-budget: cost-test $(PY) $(TOOLS)/cb-cost.py --shape-budget # CB-01/CB-02: live spend since the last commit. cost-budget: cost-test $(PY) $(TOOLS)/cb-cost.py --budget # CB-RES-0003 baseline: mechanical vs judgment turns. cost-mix: cost-test $(PY) $(TOOLS)/cb-cost.py --composition cost-pin: cost-test $(PY) $(TOOLS)/cb-cost.py --pin fc76445 --composition --by-task sim: $(IN_REPO) $(CARGO) run -q -p cb-sim -- $(REPO)/scenarios/ground/*.yaml ## Criterion benches (AM-6/AM-7) bench: $(IN_REPO) $(CARGO) bench -p games-ground ## InnerLoop positive control: run every bench once, no measurement. ## Fails if a workload stalls or produces the wrong event count. bench-test: $(IN_REPO) $(CARGO) bench -p games-ground --bench synthetic -- --test ## AM-2/AM-3 input: source LOC per crate (excludes tests would need tokei) loc: @$(IN_REPO) for d in crates/cb-kernel crates/cb-events crates/cb-game-runtime games/ground tools/cb-sim; do \ printf '%-28s %s\n' $$d "$$(find $$d/src -name '*.rs' | xargs cat | grep -vcE '^\s*(//|$$)')"; \ done all: check test sim coverage size-metrics runtime-metrics am6 am7 am8 edition-check replay-test dep-weight self-tests env-test facts-check loop-lint bench-test