CB-WP-0007 T01+T03: window the metric, budget it, cap meta at 25%
Scope cut first, on the maintainer's decision after a spend review: the project is 38% product / 62% loop-meta, cost per response is 2.9x worse than its best window, and INTENT stage 0 still lacks a CLI player and bots. CB-WP-0005 and CB-WP-0006 cost ~$74 — 31% of all spend — for zero measured efficiency gain. T02 and T04 are cancelled unstarted. T01: SH-1/SH-2/SH-3 now report over the window since the last commit, and the cumulative figure is retained but labelled "history, NOT the metric". The prediction held decisively — window 655,744 mean context against cumulative 255,307, a 2.6x gap against a 20% refutation threshold. A cumulative mean over 1,094 responses cannot detect a worsening trend because the history outvotes the present. T03: `make shape-budget`, modelled on CB-01/CB-02. Soft thresholds are the existing SessionShape targets; hard is 1.5x, set before the next measurement per §Step 4. Deliberately not in `make all` — failing the build on context would block committing, and committing is what closes the attribution window and is the natural point to compact, so a gate that blocks the remedy is a trap. It fires HARD on its first run: 656,574 against a 300,000 ceiling. InnerLoop v1.5 establishes the soft 25% meta budget. Workplans declare kind: product|meta|mixed and `make status` reports the share; mixed splits 50/50 and says so. Soft on purpose — a task already started may be finished, because stopping mid-task to satisfy a ratio wastes the work. What it forbids is opening new meta work above the line. A pass that exceeds it must say so in its evidence and name the product work displaced. First reading: 68% OVER, of $74.22 attributed. Product reads $0.00 because the only product workplan, CB-WP-0001, predates qualified task ids and its bare T## labels collide across passes — stated in the output rather than papered over. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
98e09c4394
commit
b79ea9690d
11 changed files with 229 additions and 26 deletions
118
tools/cb-cost.py
118
tools/cb-cost.py
|
|
@ -396,7 +396,8 @@ def collect(slug, pin_ref=None, since_ref=None):
|
|||
for r in responses:
|
||||
r["cost"] = price_of(prices, r["model"], r["toks"], r["timestamp"])
|
||||
|
||||
attribute(responses, commit_index(pin))
|
||||
commits = commit_index(pin)
|
||||
attribute(responses, commits)
|
||||
|
||||
by_task = collections.defaultdict(float)
|
||||
# T02: per-task token detail, so a task close can report *measured*
|
||||
|
|
@ -454,9 +455,28 @@ def collect(slug, pin_ref=None, since_ref=None):
|
|||
mech = [r for r in responses
|
||||
if set(r.get("categories") or []) & MECHANICAL]
|
||||
|
||||
# The window the SH-* targets compare against: responses since the
|
||||
# last commit when unpinned, or the whole pinned range when pinned
|
||||
# (a pin is already a window someone chose deliberately).
|
||||
if pin or since:
|
||||
window_responses, window_label = responses, "the pinned/--since range"
|
||||
else:
|
||||
last = commits[-1][0] if commits else ""
|
||||
window_responses = [r for r in responses if last and r["timestamp"] > last]
|
||||
window_label = "since the last commit"
|
||||
|
||||
sub = sum(r["cost"] or 0 for r in responses if r["subagent"])
|
||||
return {
|
||||
"session_shape": session_shape(responses),
|
||||
# CB-WP-0007 T01. SH-1/SH-2 as a cumulative mean over every
|
||||
# response ever recorded cannot detect a worsening trend: the
|
||||
# history outvotes the present. CB-WP-0006 ran at 503,464 mean
|
||||
# context and the cumulative figure still printed 206,952.
|
||||
# Both are reported; only the window is the metric.
|
||||
"session_shape_window": session_shape(window_responses)
|
||||
if window_responses else None,
|
||||
"window_responses": len(window_responses),
|
||||
"window_label": window_label,
|
||||
"tool_mix": {
|
||||
"turns": dict(mix_turns),
|
||||
"cost": dict(mix_cost),
|
||||
|
|
@ -550,16 +570,33 @@ def render(rep, by_task=False, composition=False):
|
|||
f"{mix['mechanical_turns']:>5} turns ${mix['mechanical_cost']:>8,.2f}"
|
||||
f" = {100*mix['mechanical_cost']/(rep['total'] or 1):.0f}% of pass")
|
||||
|
||||
def _shape_lines(sh, indent=" "):
|
||||
print(f"{indent}SH-1 mean context {sh['SH-1_mean_context']:>12,.0f} tok "
|
||||
f"[{'ok ' if sh['SH-1_mean_context']<=200_000 else 'FAIL'} target 200,000]")
|
||||
print(f"{indent}SH-2 p90 context {sh['SH-2_p90_context']:>12,.0f} tok "
|
||||
f"[{'ok ' if sh['SH-2_p90_context']<=300_000 else 'FAIL'} target 300,000]")
|
||||
print(f"{indent}SH-3 batching rate {100*sh['SH-3_batching_rate']:>11.1f}% "
|
||||
f"[{'ok ' if sh['SH-3_batching_rate']>=0.20 else 'FAIL'} target 20.0%]")
|
||||
print(f"{indent} {sh['tool_calls']} tool calls in "
|
||||
f"{sh['responses_with_tools']} responses; "
|
||||
f"{sh['calls_in_batched_turns']} in batched turns")
|
||||
|
||||
win = rep.get("session_shape_window")
|
||||
print(f"\n session shape (specs/SessionShape.md) — window: "
|
||||
f"{rep.get('window_label', '?')}")
|
||||
if win:
|
||||
print(f" THE METRIC — {rep['window_responses']} response(s)")
|
||||
_shape_lines(win)
|
||||
else:
|
||||
print(" THE METRIC — no responses in the window "
|
||||
"(nothing since the last commit)")
|
||||
|
||||
sh = rep["session_shape"]
|
||||
print("\n session shape (specs/SessionShape.md)")
|
||||
print(f" SH-1 mean context {sh['SH-1_mean_context']:>12,.0f} tok "
|
||||
f"[{'ok ' if sh['SH-1_mean_context']<=200_000 else 'FAIL'} target 200,000]")
|
||||
print(f" SH-2 p90 context {sh['SH-2_p90_context']:>12,.0f} tok "
|
||||
f"[{'ok ' if sh['SH-2_p90_context']<=300_000 else 'FAIL'} target 300,000]")
|
||||
print(f" SH-3 batching rate {100*sh['SH-3_batching_rate']:>11.1f}% "
|
||||
f"[{'ok ' if sh['SH-3_batching_rate']>=0.20 else 'FAIL'} target 20.0%]")
|
||||
print(f" {sh['tool_calls']} tool calls in {sh['responses_with_tools']} responses; "
|
||||
f"{sh['calls_in_batched_turns']} in batched turns")
|
||||
print(f"\n history, {rep['responses']} responses — context only, NOT the "
|
||||
f"metric.")
|
||||
print(" A cumulative mean cannot detect a worsening trend; the history")
|
||||
print(" outvotes the present (CB-RES-0005 §1).")
|
||||
_shape_lines(sh)
|
||||
|
||||
if rep["unpriced"]:
|
||||
print(f"\n UNPRICED ({len(rep['unpriced'])} responses, model not in sheet):")
|
||||
|
|
@ -690,6 +727,63 @@ def self_test():
|
|||
return 0 if ok else 1
|
||||
|
||||
|
||||
# CB-WP-0007 T03 / CB-RES-0005 D2. Soft = the SessionShape target; hard is
|
||||
# 1.5x, set here BEFORE the next measurement per InnerLoop §Step 4.
|
||||
# Reported, never in `make all`: failing the build on context would block
|
||||
# committing, and committing is what closes the attribution window and is
|
||||
# the natural point to compact. A gate that blocks the remedy is a trap.
|
||||
SHAPE_SOFT = {"SH-1": 200_000, "SH-2": 300_000}
|
||||
SHAPE_HARD = {"SH-1": 300_000, "SH-2": 450_000}
|
||||
|
||||
|
||||
def shape_budget(slug):
|
||||
"""SH-* for the window since the last commit, with soft/hard verdicts."""
|
||||
try:
|
||||
rep = collect(slug, None)
|
||||
except Abort as e:
|
||||
print(f"ABORT — {e}", file=sys.stderr)
|
||||
return 1
|
||||
win = rep.get("session_shape_window")
|
||||
head = subprocess.run(
|
||||
["git", "-C", REPO, "log", "-1", "--format=%h %s"],
|
||||
capture_output=True, text=True, check=True).stdout.strip()
|
||||
|
||||
print("session-shape budget — the window since the last commit")
|
||||
print(f" last commit {head}")
|
||||
# Positive control: a budget that reports ok because it measured
|
||||
# nothing is the harness-does-nothing shape, in a reporting tool.
|
||||
if not win or rep["window_responses"] == 0:
|
||||
print(" no responses since the last commit — nothing to report",
|
||||
file=sys.stderr)
|
||||
return 1
|
||||
print(f" window {rep['window_responses']} response(s)")
|
||||
|
||||
breach = 0
|
||||
for key, label, val in (
|
||||
("SH-1", "mean context", win["SH-1_mean_context"]),
|
||||
("SH-2", "p90 context ", win["SH-2_p90_context"]),
|
||||
):
|
||||
soft, hard = SHAPE_SOFT[key], SHAPE_HARD[key]
|
||||
mark = "ok " if val <= soft else ("SOFT" if val <= hard else "HARD")
|
||||
print(f" {key} {label} {val:>10,.0f} tok [{mark}] "
|
||||
f"soft {soft:,} / hard {hard:,}")
|
||||
breach = max(breach, 0 if val <= soft else (1 if val <= hard else 2))
|
||||
rate = win["SH-3_batching_rate"]
|
||||
print(f" SH-3 batching {100*rate:>9.1f}% "
|
||||
f"[{'ok ' if rate >= 0.20 else 'SOFT'}] floor 20.0%")
|
||||
|
||||
if breach >= 2:
|
||||
print("\n HARD — compact before continuing. Context this size costs "
|
||||
"~10x per turn\n against a compacted session "
|
||||
"(specs/SessionShape.md §2).", file=sys.stderr)
|
||||
return 1
|
||||
if breach == 1:
|
||||
print("\n SOFT — over target. Compaction is the remedy and it is free.")
|
||||
else:
|
||||
print("\n within budget")
|
||||
return 0
|
||||
|
||||
|
||||
def budget(slug, soft, hard):
|
||||
"""Live cost budget (specs/CostAccounting.md §7).
|
||||
|
||||
|
|
@ -734,6 +828,8 @@ def main():
|
|||
ap.add_argument("--composition", action="store_true")
|
||||
ap.add_argument("--session-shape", action="store_true",
|
||||
help="SH-1..SH-3 (always shown in the default report)")
|
||||
ap.add_argument("--shape-budget", action="store_true",
|
||||
help="SH-1/SH-2/SH-3 for the window since the last commit")
|
||||
ap.add_argument("--budget", action="store_true",
|
||||
help="CB-01/CB-02: spend since the last commit, live")
|
||||
ap.add_argument("--soft", type=float, default=10.00)
|
||||
|
|
@ -745,6 +841,8 @@ def main():
|
|||
if args.self_test:
|
||||
return self_test()
|
||||
|
||||
if args.shape_budget:
|
||||
return shape_budget(args.slug)
|
||||
if args.budget:
|
||||
return budget(args.slug, args.soft, args.hard)
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue