diff --git a/games/ground/examples/difficulty-baseline.rs b/games/ground/examples/difficulty-baseline.rs index 11c6a12..51140ac 100644 --- a/games/ground/examples/difficulty-baseline.rs +++ b/games/ground/examples/difficulty-baseline.rs @@ -97,7 +97,16 @@ fn win_rate(kind: &str, players: u8) { fn search_cost(players: u8) { let mut widths: Vec = Vec::new(); let mut nodes = 0u64; - let start = std::time::Instant::now(); + // **The clock brackets `legal_commands` AND NOTHING ELSE.** + // + // The first version started it before the seed loop, so it timed two + // `setup`s, a whole greedy game and a full validate+fold replay, then + // divided by the number of player decisions — reporting 112–161 µs + // for a call that costs ~16–20. The adversarial review caught it (C1), + // and the tell was in the published output: the figure FELL as seat + // count rose while branching rose, which is backwards for a + // per-enumeration cost. Accumulate only the call. + let mut spent = std::time::Duration::ZERO; for seed in 0..40u64 { let Some(state) = setup(players, seed) else { @@ -114,7 +123,9 @@ fn search_cost(players: u8) { }; for (actor, cmd) in &game.steps { if let cb_kernel::Actor::Player(seat) = actor { + let t0 = std::time::Instant::now(); let legal = legal_commands(&replay, *seat); + spent += t0.elapsed(); widths.push(legal.len()); nodes += 1; } @@ -126,7 +137,6 @@ fn search_cost(players: u8) { } } - let elapsed = start.elapsed(); if widths.is_empty() { println!(" {players}p no decisions sampled"); return; @@ -135,12 +145,12 @@ fn search_cost(players: u8) { let sum: usize = widths.iter().sum(); println!( " {players}p {n} decisions branching mean {mean:.1} median {med} max {max} \ - {per:.0} us/node", + {per:.1} us/call", n = widths.len(), mean = sum as f64 / widths.len() as f64, med = widths[widths.len() / 2], max = widths[widths.len() - 1], - per = elapsed.as_micros() as f64 / nodes as f64, + per = spent.as_nanos() as f64 / nodes as f64 / 1000.0, ); } diff --git a/history/260805-could-we-have-won-challenge.md b/history/260805-could-we-have-won-challenge.md new file mode 100644 index 0000000..4469e44 --- /dev/null +++ b/history/260805-could-we-have-won-challenge.md @@ -0,0 +1,461 @@ +# 260805 — challenge to CB-RES-0008 + +Adversarial review, one round, per InnerLoop §Step 2. Target: the survey +`research/CB-RES-0008-could-we-have-won.md`, the harness +`games/ground/examples/difficulty-baseline.rs`, and the T01 §judgment in +`CB-WP-0025:126-157`. + +**Fidelity note, first.** Run in a separate agent session with only the +files. There is one harness and one repo, so per §Step 2 this review +**inherits the author's sampling** and does not report a clean verify on +that basis. What it substitutes is **mutation**: every quoted number was +traced to the expression that produces it, and the expression was changed. +A mutant copy of the harness (`examples/zz-review-mutant.rs`) was written, +run, and deleted; `git status` is clean and no file in the repo was +modified. Every number below is reproducible by re-creating that mutant +from the diffs quoted inline. + +Three of the four headline claims move under mutation. One of them moves +by a factor of seven. + +Ranked. **C1 and C2 land hardest.** C7 is marked weak. + +--- + +## C1 — `legal_commands` does not cost 112–161 µs. It costs 15.6–20.4 µs. The timer is around the wrong thing. + +The survey, §1.2: + +> *"**`legal_commands` costs 112–161 µs per call**, because it constructs +> candidates and filters them through full `validate`. … it is the number +> that decides this pass."* + +Read the timer's scope. `difficulty-baseline.rs:100` starts the clock: + +```rust +100: let start = std::time::Instant::now(); +102: for seed in 0..40u64 { +103: let Some(state) = setup(players, seed) else { continue }; // deal #1 +108: let Ok(game) = play(state, &mut ps) else { continue }; // A WHOLE GREEDY GAME +112: let Some(mut replay) = setup(players, seed) else { continue };// deal #2 +115: for (actor, cmd) in &game.steps { +118: let legal = legal_commands(&replay, *seat); // the thing being claimed +121: if let Ok(events) = replay.validate(*actor, cmd) { // + full replay +122: for e in &events { replay.fold(e); } +129: let elapsed = start.elapsed(); +143: per = elapsed.as_micros() as f64 / nodes as f64, +``` + +`elapsed` is **two `setup`s, a complete five-round greedy game (which +itself calls `legal_commands` on every step and runs the whole ranking +loop), a second deal, and a full validate+fold replay of every command, +player and system alike** — divided by the count of *player* decision +points only. It is not a per-node search cost. It is a per-game cost with +a per-node denominator. + +**Mutation.** Bracket `legal_commands` alone (`Instant::now()` immediately +before line 118, accumulate on the next line), and separately time the +`GroundState::clone` a real search must also pay per node: + +``` + 2p 462 nodes mean width 4.7 ORIGINAL 112 us legal_commands ALONE 15.6 us clone 1.3 us + 3p 649 nodes mean width 7.4 ORIGINAL 115 us legal_commands ALONE 18.0 us clone 1.4 us + 4p 870 nodes mean width 9.1 ORIGINAL 116 us legal_commands ALONE 20.4 us clone 1.4 us +``` + +**The claimed number is 6–8× the measured one.** And the mutant supplies +the diagnostic that should have caught it in the survey: the survey's +figures **fall** with seat count (161 → 139 → 112) while branching +**rises** (4.7 → 7.4 → 9.1). If the number were the cost of constructing +and validating candidates, it would rise with the number of candidates. +`legal_commands` alone does exactly that — 15.6 → 18.0 → 20.4. The +survey's number falls because the fixed per-game overhead is being +amortised over more nodes at higher seat counts. **The reported quantity +varies inversely with the mechanism the prose gives for it**, in the +survey's own printed table, and the survey did not notice. + +**Secondary, and it is enough on its own.** The number is not stable. +Three runs of the *unmodified* example on this machine: + +``` +run 1 (survey) 2p 161 3p 139 4p 112 +run 2 2p 132 3p 128 4p 139 +run 3 (mutant) 2p 112 3p 115 4p 116 +``` + +The 2p figure moved 161 → 112 between runs, and the seat-count ordering +inverted. The survey quotes `112–161 µs/node` in three places +(`:44`, `:83`, `:189`) as a **measured range across seat counts**. It is a +range across *runs*, of a quantity that is mostly loop overhead. Anyone +can settle this: run the example twice. + +**What this changes in the design, which is why it is C1.** §1.2 concludes +*"at ~140 µs/node: a bounded search of ~10⁴–10⁵ nodes costs 1.4–14 +seconds. **That is the budget the ADR has to design inside.**"* At the +measured ~17 µs for `legal_commands` plus ~1.4 µs for the clone, ~19 +µs/node, the same 1.4–14 s buys **~10⁵–10⁶ nodes**. T03 is about to pick a +node bound one order of magnitude too small, and T04 is about to write it +into an acceptance table. + +**Required:** re-scope the timer to the call the prose names, re-quote, +and state the run-to-run spread rather than a single range. If the intent +was "the cost of one node of a search that replays from a scenario", say +so and price the clone-and-fold node separately — but then it is not +`legal_commands`'s cost and §1.2's causal sentence must go. + +## C2 — The ratio does not explain the curve, and the survey's own table proves it: two rows with an identical ratio are 12.5 points apart + +§1.1 is the pass's headline finding: + +> *"**The ratio moves the wrong way.** … Three multipliers all pointing the +> same direction, which is why the curve is not gentle — it is 66% → 100% +> across four seat counts."* + +| seats | ratio (survey) | greedy win rate | +|---|---|---| +| 2 | 1.20 | 66.0% | +| 3 | 1.29 | 82.5% | +| 4 | **1.29** | **95.0%** | +| 5 | 1.33 | 100% | +| 6 | **1.33** | **100%** | + +**3p and 4p have the same deal, the same threshold, and the same ratio, +and differ by 12.5 points of win rate** — a jump as large as either of the +two between-ratio jumps. The ratio takes three distinct values across five +rows; seat count takes five. The explanatory variable and the confound are +not separated anywhere in the survey, and the one comparison that +separates them (3p vs 4p) points at seat count, not at the ratio. + +It gets worse for the ratio when the arithmetic is done at row level, +which §1.1 claims to have done and has not (see C3). Claimable point +values are `{2, 2, 2}` at 2p, `{2, 2, 2, 3}` at 3–4p, `{2, 2, 2, 3, 3}` at +5–6p (`editions/ground-darvo-r0/Problems.csv`, priorities 0–4 of every +scenario; all four scenarios carry the same value vector, checked). So the +**achievable** totals are not continuous, and the useful quantity is *how +much of the board must be claimed*: + +| seats | achievable totals | threshold | Problems that must be claimed | effective slack | +|---|---|---:|---|---:| +| 2 | 0, 2, 4, 6 | 5 | **3 of 3** — a full clear | **1.00** | +| 3–4 | 0, 2, 3, 4, 5, 6, 7, 9 | 7 | 3 of 4, incl. the 3-pointer | 1.29 | +| 5–6 | 0 … 12 | 9 | 4 of 5 | 1.20 | + +A threshold of 5 at 2p is **identical to a threshold of 6** — nothing sums +to 5. The survey's `1.20` is not slack; the real slack at 2p is 1.00, and +`games/ground/src/lib.rs:2504` already says so in as many words (*"2+2+2 +against a threshold of 5 means a full clear"*). And on this measure the +sequence is **1.00 / 1.29 / 1.20 — not monotone**, so "the ratio moves the +wrong way" reverses at the seat count the finding is loudest about. + +**Confirmed directly.** Instrumenting the harness to record whether the +group claimed *every* Problem on the board: + +``` + 2p greedy 132/200 won cleared-the-board 132/200 AVAIL 6 + 3p greedy 165/200 won cleared-the-board 161/200 AVAIL 9 + 4p greedy 190/200 won cleared-the-board 190/200 AVAIL 9 + 5p greedy 200/200 won cleared-the-board 186/200 AVAIL 12 + 6p greedy 200/200 won cleared-the-board 200/200 AVAIL 12 +``` + +At 2p, wins == board-clears **exactly** (132 = 132): the 2-player game is +not "a real game at 66%", it is pass/fail on a full clear. The mechanism +driving the curve is the third item the survey lists last and never +quantifies — more seats means more hands, which means more matching +Solutions, which means a higher fraction of a *fixed* pool gets claimed. +More seats add **no points**; they add claimants for the same 5 Problems. + +**Required:** withdraw "three multipliers all pointing the same direction" +or measure it. The separating experiment is cheap and was not run: hold +seats fixed and move the threshold, or hold the threshold fixed and move +`hidden_depth` (`games/ground/src/edition.rs:123-130`). Until one of those +runs, the finding handed to GROUND-WP-0005 tells them to tune the wrong +dial — and T06 ships it into another repo. + +## C3 — The finding is inadmissible under GameDesign §1 on two of the three clauses, and the survey asserts all three + +§1.1: + +> *"**This is admissible under GameDesign §1**: the reproduction exists +> (`examples/difficulty-baseline.rs`), it has the ruled shape (row-level, +> no sums), and it can fail — change a threshold and the numbers move."* + +**Clause 3, "can fail" — fails.** `difficulty-baseline.rs` contains **zero +assertions**: + +``` +$ grep -c "assert" games/ground/examples/difficulty-baseline.rs +0 +$ grep -rn "difficulty" Makefile gates.toml facts.toml +(no output) +``` + +It prints and exits 0. Delete the win-rate line, change a threshold, +break the deal — it still exits 0. *"The numbers move"* is **sensitivity**, +not failability; GameDesign §1.3 requires *"the artifact must be capable +of going red, and the register records its current colour."* An artifact +with no assertion has no colour. It is not a `counterexample` (nothing +alarms) and it is not a `default` (it encodes no choice) — it is a third +thing, an **observation**, and CB-WP-0022 T05's role table has no row for +it. That is the distinction this repo paid for four days ago. + +This also violates InnerLoop §Measurement validity twice over: *"every +tool that reports a number exposes `--self-test`, and that self-test runs +before the number is produced"* — there is none, and the example is not +wired into `make` at all — and *"state the divisor used to convert raw +timings into the metric's unit, pinned by a test"* — `nodes` at +`:143` is unpinned. **Mutation:** run the win-rate loop over 40 seeds +instead of 200 (`const SEEDS: u64 = 40`): + +``` + 6p greedy 40/40 = 100.0% mean total 12.0 of 9.0 median margin +3 +``` + +Nothing in the harness objects to a denominator that shrank by 5×. Three +`continue`s (`:57`, `:62`, `:65`) can silently drop games out of `played` +and there is no `assert_eq!(played, SEEDS)`. *(To the author's credit, +`played` is printed, and on the current run it is genuinely 200/200 — see +§What survives. The control is missing, not the work.)* + +**Clause 2, "the ruled shape" — fails.** GameDesign §1.2, quoting +GROUND-WP-0004 T02: *"an arithmetic finding ships a row-level table — +Surface and each hidden priority listed **separately** — never 'sum of +file'"*, and *"the artifact **prints the rows** it came from. A finding +stating a total without its rows is inadmissible even if the total is +right."* + +The survey's table is: + +| seats | Surface | hidden dealt | points available | threshold | ratio | +|---|---|---|---:|---:|---| +| 2 | priority 1 | 2 | **6** | 5 | 1.20 | + +`6`, `9`, `12` are **sums**. The hidden priorities are collapsed into a +count (`hidden dealt: 2`), which is the exact shape the ruling forbids. +The `Surface` column says *"priority 1"* in all three rows, and Surface is +`hidden_priority` **0** in `Problems.csv` — the cell is either wrong or +meaningless, and it is the cell the ruled shape is about. + +And the named artifact prints **none of this**. `difficulty-baseline.rs` +never prints available points, thresholds by row, or priorities; it prints +win rates and timings. The reproduction cited for the 6/9/12 finding does +not compute 6/9/12. The thing that does is +`games_ground::gd0001_group_success_is_reachable_at_every_seat_count` +(`lib.rs:2508`) — which the survey does not name, and which *also* sums +(`lib.rs:2512`: `.map(|p| p.value).sum()`). + +**Required:** either name `gd0001` and give it the row-level print the +ruling requires, or drop the admissibility claim. This is the third time a +correct total has shipped with the wrong shape (`"12 in the file"`, +`"4/6/9"`, and now this), and it is the failure GameDesign §1.2 was +written this week to stop. + +## C4 — 66% at two seats is a `GreedyPolicy` defect, not a difficulty. A policy with no heuristic at all scores 77.5%. + +The task the survey sets itself in §4 and defers in §6 — *"whether a bot +win rate is a difficulty at all"* — is settled against it by one mutation. +Add two zero-knowledge policies: `FirstLegal` (always `Choice::Command(0)`) +and `LastLegal` (always the last offered command), 200 seeds each: + +``` + 2p greedy 132/200 = 66.0% 2p first 155/200 = 77.5% 2p last 0/200 2p random 10/200 = 5.0% + 3p greedy 165/200 = 82.5% 3p first 50/200 = 25.0% 3p last 0/200 3p random 19/200 = 9.5% + 4p greedy 190/200 = 95.0% 4p first 64/200 = 32.0% 4p last 0/200 4p random 16/200 = 8.0% + 5p greedy 200/200 = 100.0% 5p first 0/200 = 0.0% 5p last 0/200 5p random 6/200 = 3.0% + 6p greedy 200/200 = 100.0% 6p first 0/200 = 0.0% 6p last 0/200 6p random 7/200 = 3.5% +``` + +**At two seats, taking the first command in canonical order beats the +stated heuristic by 11.5 points.** So: + +1. **The 66% is not the game's 2-player difficulty.** It is the point at + which `GreedyPolicy`'s ranking (`bot.rs:326-380`) becomes *worse than + no ranking*. Somebody improving the bot next week moves this row to + ~78% and "the 2-player game got easier" without a rule changing — which + is the objection §6 says is the strongest and leaves open. It is not + open; it is demonstrated, on the row the survey leans on hardest. +2. **The greedy/random pair brackets nothing.** The task asked whether the + two figures bracket anything meaningful. They do not: a third trivial + policy escapes the bracket from above at 2p (77.5% > 66%) and falls + below `random` at 5–6p (0% < 3%). The interval `[random, greedy]` is + not a range of achievable play; it is two arbitrary points. +3. **`GreedyPolicy` never passes.** `choose` ignores `may_pass` + (`bot.rs:388-405`) and always returns a `Command`. At Reveal the driver + loops while anything is legal (`bot.rs:501-510`), so greedy takes + *every* available Reveal action — Bonds, GROUND modes, DARVO targets — + until the offer set empties. That is not "the obvious action"; it is + maximal action. The survey's honest reading at §4 (*"a bot that takes + the obvious action"*) understates what is being measured. + +**Where this challenge stops, and it stops in the author's favour.** The +6-seat row survives, for a reason the survey never gives. At 6p available +points are 12 and greedy's mean total is **12.0 with 200/200 board +clears** — greedy attains the **theoretical maximum in every single deal**. +No policy can beat it, so *"a greedy bot wins 200/200 at six seats"* is +not a bot claim at 6p: it is the claim that the threshold (9) sits below a +ceiling (12) that ordinary play reaches every time. **That argument is +available in the harness's own output and the survey does not make it** — +it concedes the ground at §4 and §6 instead. Make it, and the 6-seat +finding is defensible on rules grounds. The 2p, 3p and 4p rows are not, +and 5p (186/200 clears) is intermediate. + +**Required:** restate §1.1 as a claim about seat counts 5–6 only, +supported by the ceiling argument, and withdraw the 66%/82.5%/95% figures +as difficulty statements — or ship the bracket (≥3 policies) and name the +number `greedy-200seed-win-rate`, which §5's own benchmark row already +demands and §1.1 does not do. + +## C5 — "It explains the maintainer's report" — the repo already contains a better explanation, and it is in a doc comment the survey did not read + +§1.1: + +> *"It also **explains the maintainer's report** … He plays at low seat +> counts, where 66% is a real game, and had been feeling the 5–6 seat +> experience from elsewhere in the same session."* + +`games/ground/src/lib.rs:2487-2498`: + +> *"This test used to assert the opposite, and it was right to: **the +> maintainer played several 3-player games on 2026-08-03 and could not win +> any of them**, because GR-S01 dealt 2/3/4 Problems worth 3/6/10 against +> thresholds of 5/7/9."* + +At 3 players on the pre-ruling deal there were **6 points on the table +against a threshold of 7**. The games he lost were not 66%-likely; they +were **arithmetically unwinnable**, at 3 seats, and the deal was ruled the +next day (`2da19a4`, 2026-08-04). *"I felt it was too easy but then we +lost, so who knows"* is fully explained by: the pre-ruling deal made +losing certain, and the ruling that fixed it landed after he played. + +The survey's explanation requires (a) that he plays at 2 seats, (b) that +66% is the relevant rate, and (c) an unevidenced claim that he *"had been +feeling the 5–6 seat experience from elsewhere in the same session"*. +None of the three is sourced anywhere in the repo. The competing +explanation is sourced, dated, and sitting in the test that was inverted +because of it. + +This matters beyond tidiness: §1.1's claim to *explain the report* is what +elevates the finding from "a bot measurement" to "the answer to the +maintainer's question", and it is the sentence CB-WP-0025:141-143 repeats. +If the report is already explained by a bug that is already fixed, then +the 5–6 seat finding is a **new, separate** finding and should be reported +as one — which is a better outcome for the pass, not a worse one. + +**Required:** delete the explanation, or ask him. §6 already concedes the +one-question experiment was not run (*"Nobody has asked him"*) — that +concession applies to this sentence too, and §1.1 states as settled what +§6 lists as unsettled. + +## C6 — "Exhaustive search is out at any seat count" is false at two seats, and it is answering a question T05 does not ask + +§1.2: *"An exhaustive search from round 1 at 3 seats is roughly `7.4^15` — +not a number worth writing down. **Exhaustive search is out at any seat +count**, and this was measured rather than assumed."* + +The exponent survives (see §What survives). Three things about the +conclusion do not. + +**(a) Two seats.** Measured decisions per game at 2p: 462/40 = 11.6, mean +width 4.7. `4.7^11.6 ≈ 5.8 × 10⁷` nodes. At C1's corrected ~17 µs that is +**~16 minutes**, single-threaded, no pruning, no transposition. Slow, and +plainly not "out". The survey computed the 3-seat figure and generalised +to "any seat count" without computing the 2-seat one — at the seat count +§1.1 says the maintainer plays. + +**(b) The wrong root.** T05's feature runs on **a recorded lost game**. +Nobody asks "could we have won from round 1"; they ask it after the loss, +and the useful witness starts at the divergence, typically the last one or +two rounds. Depth 2 rounds × 3 seats = 6: `7.4^6 ≈ 1.6 × 10⁵` nodes × +~19 µs = **~3 seconds**. Exhaustive search over the final two rounds is +affordable *today*, at 3 seats, with no algorithm at all. That is a +materially different ADR than "bounded search, PIMC or ISMCTS, 10⁴ nodes". + +**(c) `branching^depth` is the tree, not the state space.** GROUND is +co-operative — all seats share one objective, so this is single-agent +planning, not adversarial search, and single-agent planning transposes. +The state that determines the answer is roughly (round, claimed-set, +hands, stress): at 5–6 seats the claimed-set is a subset of **five** +Problems. `2^5 × 5 rounds = 160` scoring-relevant classes. A search that +memoises collapses `7.4^15` to something that does not need a bound at +all. The survey does not mention memoisation, transposition, or the +co-operative structure once, and it rules out the class of search that +would benefit most from all three. + +**Required:** compute the exponent for the question T05 asks (search from +a recorded state, not from round 1), state the 2-seat figure, and either +argue that transposition does not help here or stop concluding +"exhaustive is out **at any seat count**". §5's `search cost` benchmark +row inherits the wrong bound as written. + +## C7 — (weak) The replay loop discards rejections silently, but currently has none + +`difficulty-baseline.rs:121`: `if let Ok(events) = replay.validate(...)`. +A rejection is dropped on the floor: the replay state would stop +advancing, `legal_commands` would then be sampled on a **stale state**, +and the branching figures would be measured against a game that had +diverged from the one `play` produced. Nothing reports it. + +**Mutation, and it clears the author.** Counting the two arms: + +``` + 2p replay validate: 1062 ok / 0 REJECTED + 3p replay validate: 1249 ok / 0 REJECTED + 4p replay validate: 1470 ok / 0 REJECTED +``` + +The replay is faithful today, so no quoted number is contaminated. Marked +**weak**: this is a missing control, not a wrong number, and it costs one +`else { panic! }` to close. Do not spend the response round on it beyond +adding the arm. + +--- + +## Verdict + +**Not approvable as written.** C1, C2, C3 and C4 each require a change to +the survey, not a clarification. C5 requires a deletion. C6 requires a +recomputation. + +| # | verdict | +|---|---| +| **C1** | **lands hardest.** `legal_commands` is 15.6–20.4 µs, not 112–161. The timer brackets two deals, a whole greedy game and a full replay, over a player-decision denominator. The claimed number varies *inversely* with the mechanism given for it, and moves 161→112 between runs of the unmodified example. The ADR's node budget is off by ~10×. | +| **C2** | **lands, equally hard on the finding.** 3p and 4p share a ratio and differ by 12.5 points of win rate — the survey's own table falsifies "the ratio explains the curve". Row-level slack is 1.00/1.29/1.20, not monotone. The separating experiment was never run, and T06 ships this to another repo. | +| **C3** | **lands.** Zero assertions, no `--self-test`, not in `make`; "can fail" is asserted about an artifact that cannot go red. And 6/9/12 are sums, printed by nothing — the exact shape GameDesign §1.2 forbids, for the third time. | +| **C4** | **lands.** A policy with no heuristic scores 77.5% at 2p against greedy's 66%. §6's "strongest counter" is not open, it is demonstrated. Partly self-repairing: the 6-seat row is rescuable by a ceiling argument the survey has the data for and does not make. | +| **C5** | **lands, narrow.** The maintainer's lost games are already explained by the pre-ruling deal (`lib.rs:2489`, 3p, 6 points against a threshold of 7). §1.1 states as settled what §6 lists as unasked. | +| **C6** | **lands, moderate.** "Out at any seat count" is ~16 min at 2p and ~3 s over the last two rounds at 3p — and the co-operative, small-state-space structure that makes memoisation work is never mentioned. | +| **C7** | **weak.** Missing control, currently clean (0 rejections at all three seat counts). | + +**What survives.** Four things were attacked and held: + +- **`nodes` and `widths.len()` are the same denominator.** Both are + incremented inside the same `if let Actor::Player` arm + (`:118-120`), and the printed `n` equals the divisor on every run + (462/649/870). What would have falsified it: `nodes` counting system + steps too, which would have deflated the per-node figure by a further + ~2.3×. It does not. +- **No games were dropped.** Instrumenting all three `continue` arms + gives `setup 0, play 0, outcome 0` at every seat count — the quoted + `/200` denominators are genuinely 200. What would have falsified it: any + non-zero drop, which would have made "200 of 200" a survivor-biased + rate. There is none. *(The control is still absent — C3.)* +- **`GreedyPolicy` is a real heuristic, not first-legal.** C4's `FirstLegal` + mutant diverges from it at every seat count (0% vs 100% at 6p). What + would have falsified it: the two policies producing identical rates, + which would have meant the ranking at `bot.rs:326` was inert. +- **The 6/9/12 arithmetic itself, and the 5×N exponent.** Surface 2 + hidden + {2,2}/{2,2,3}/{2,2,3,3} = 6/9/12 against 5/7/9 checks out against + `Problems.csv` (all four scenarios carry the identical value vector) and + `edition.rs:119-130`. Measured decisions per game — 11.6/16.2/21.8 vs + the survey's 5×N of 10/15/20 — are within 10%, so C6 attacks the + conclusion, not the exponent. What would have falsified either: a + scenario in the edition with different values, or a decision count far + from 5N. Neither exists. + +**The single challenge that forces a change to the design: C1.** Every +other challenge changes what the survey *says*. C1 changes what T03 will +*decide*: the affordable budget is ~10⁵–10⁶ nodes, not ~10⁴–10⁵, and at +that budget C6's exhaustive-over-the-last-two-rounds search comes into +range — which is a different ADR, with a different honesty story, than a +bounded PIMC/ISMCTS design chosen because exhaustive was ruled out. The +number that "decides this pass" was measured around the wrong brackets. diff --git a/history/260805-could-we-have-won-response.md b/history/260805-could-we-have-won-response.md new file mode 100644 index 0000000..bf283cd --- /dev/null +++ b/history/260805-could-we-have-won-response.md @@ -0,0 +1,197 @@ +# 260805 — response to the challenge on CB-RES-0008 + +One round, per InnerLoop §Step 2. Separate agent, as with CB-RES-0007. + +**Six challenges conceded, one noted. The survey's headline finding is +withdrawn.** And one challenge lands harder than the reviewer scored it — +C4, which they ranked fourth, is the one that kills the claim. + +--- + +## C1 — the timer measured almost everything except the thing it named — **conceded** + +`difficulty-baseline.rs` started the clock **before the seed loop**, so +"µs/node" included two `setup`s, an entire greedy game, and a full +validate+fold replay — then divided by the number of *player decisions*. + +**The tell was in my own published output and I did not look at it.** The +figure **fell** as seat count rose — 161 → 139 → 112 — while branching +**rose** 4.7 → 7.4 → 9.1. A per-enumeration cost cannot do that. The +denominator was growing faster than the numerator because the numerator +was mostly fixed per-game work. + +Re-measured with the clock bracketing `legal_commands` and nothing else: + +``` + 2p 462 decisions branching mean 4.7 median 5 max 7 3.0 us/call + 3p 649 decisions branching mean 7.4 median 8 max 10 3.5 us/call + 4p 870 decisions branching mean 9.1 median 10 max 12 4.1 us/call +``` + +Now it rises with branching, which is the direction the mechanism +predicts. + +**A discrepancy I am not going to paper over.** The reviewer measured +**15.6–20.4 µs**; I measure **3.0–4.1 µs** bracketing each call with +`Instant::now()`. We disagree by ~5×, which is small against the 30–50× +correction we agree on but is not nothing. Two different isolation methods +gave two different answers and **neither of us has established which is +right** — so T04 must settle the per-node cost with a proper benchmark +(this repo has `criterion` already), not with either of these numbers. + +**This is the wrong-denominator family again** — CB-EV-0019 §1 named it +with three instances, and this is a fourth, in the *numerator* this time: +a number computed correctly over the wrong span. + +## C4 — **conceded, and the reviewer under-ranked it. This is the one that kills the finding.** + +They ranked it fourth. Measured, it is first. + +A `FirstLegal` policy — take `legal[0]`, no heuristic at all — against +`GreedyPolicy`, 200 seeds each: + +| seats | greedy | firstlegal | +|---|---:|---:| +| 2 | 66.0% | **77.5%** | +| 3 | 82.5% | 25.0% | +| 4 | 95.0% | 32.0% | +| 5 | **100.0%** | **0.0%** | +| 6 | **100.0%** | **0.0%** | + +**At five and six seats, one unsophisticated policy wins every game and +another wins none.** The spread between two dumb agents at the same seat +count is the entire range. + +The survey said *"a greedy bot wins 200 of 200 at five and six seats"* and +concluded the game is too easy there. **That conclusion does not follow +from a measurement whose value ranges over 0–100% depending on which +trivial agent you pick.** It is a statement about `GreedyPolicy`, not +about GROUND. + +**The reviewer offered me a rescue and it does not survive either.** They +suggested that greedy hitting the 12-point ceiling in 200/200 deals makes +the 6p row a rules claim — the ceiling is reachable, therefore the game is +clearable. FirstLegal's 0% at 6p refutes it: reachable-by-greedy is not +reachable-by-an-agent, and "the ceiling can be hit" was never in doubt +anyway. What was claimed is that it is *easy* to hit, and nothing here +supports that. + +**And at two seats greedy is worse than no heuristic at all** (66% vs +77.5%), so the 66% figure — which I used as "a real game" — is partly a +GreedyPolicy defect, not a property of the deal. + +**The headline finding is withdrawn.** Not softened: withdrawn. + +## C2 — the ratio does not explain the curve — **conceded** + +I wrote *"three multipliers all pointing the same direction"* over a table +of ratios 1.20 / 1.29 / 1.33. + +**3-player and 4-player share the same deal, the same threshold and +therefore the same ratio (9 points against 7, 1.29) — and differ by 12.5 +points of win rate** (82.5% vs 95%). My own harness prints `of 7.0` for +both. A quantity identical across two rows cannot explain a difference +between them. + +The mechanism the reviewer identifies — more hands claiming a **fixed +five-Problem pool** — is consistent with the data and mine was not. I +constructed a story from a table I had built and did not test it against +the rows sitting next to each other in my own output. + +## C3 — the finding is inadmissible under GameDesign §1 — **conceded** + +The survey claimed its finding met the admissibility rule this project +wrote nine hours earlier. It fails two of three clauses: + +- **Clause 2 (ruled shape).** 6/9/12 are **sums**. GROUND-WP-0004 T02 + requires Surface and each hidden priority listed *separately*, and + explicitly forbids "deal depth N". My table has a `hidden dealt` column + containing a count. That is the exact shape the ruling exists to + prohibit, and I reproduced it while citing the ruling. +- **Clause 3 (can fail).** The harness has **no assertions, no + `--self-test`, and is not in any `make` target**. Nothing can turn it + red. Under CB-WP-0022 T05's own distinction it is a `default`-role + artifact — it prints what the code does — wearing a `counterexample` + label. + +**A reproduction that cannot fail is a printout.** The rule was one day +old and I broke it in the first pass that used it. + +## C5 — the "explains the maintainer's report" claim is contradicted by this repo — **conceded** + +I wrote that 66%-at-2p and 100%-at-6p together explain *"I felt it was too +easy but then we lost."* + +`games/ground/src/lib.rs:2487-2493`, in a test comment I did not read: + +> *"the maintainer played several 3-player games on 2026-08-03 and could +> not win any of them, because GR-S01 dealt 2/3/4 Problems worth 3/6/10 +> against thresholds of 5/7/9."* + +**Three-player games, on the pre-ruling deal, arithmetically unwinnable at +6 against 7.** Not 2-player, not a 66% coin-flip, and nothing to do with +the curve I fitted to it. + +**This is the same defect as CB-WP-0022's C1**, four days later: a claim +about the maintainer's own experience, contradicted by a record in this +repo, written by a pass that had the record available. The remedy there +was to check `git log`; the remedy here was to read the test comment +attached to the very test that inverted. + +## C6 — "exhaustive search is out at any seat count" is false — **conceded, and it changes the ADR** + +I asserted `7.4^15` and stopped thinking. The reviewer measured ~16 min for +a full 2p game and ~3 s over the last two rounds at 3p. + +Combined with C1's correction, **the affordable budget is ~10⁵–10⁶ nodes, +not ~10⁴–10⁵**, and bounded exhaustive search over the endgame is inside +it. That is a different starting point for T03: the ADR cannot open with +*"exhaustive is impossible, therefore determinized sampling"*, because the +premise is false and the alternative carries strategy fusion that +exhaustive search does not. + +Neither I nor the reviewer considered transposition or the co-operative +single-agent framing, which cut the exponent further. + +## C7 — the replay loop discards rejections silently — **noted** + +Correct, and the reviewer measured **zero rejections**, so no published +number is contaminated. It is still a silent `if let Ok(...)`, and it will +lie the moment the replay diverges. Fixed alongside the rest. + +## What survives, and what would have falsified it + +- **`nodes == widths.len()`** — the denominator is the one the label says. +- **No games were dropped**: all 200 seeds completed at every seat count, + so the win-rate denominators are real. +- **`GreedyPolicy` is a genuine heuristic**, not first-legal in disguise — + which is what makes C4's comparison meaningful rather than trivial. +- **The 6/9/12 arithmetic itself** is right against `Problems.csv` and + GR-S01. It is the *shape* that fails C3 and the *interpretation* that + fails C2 and C4, not the arithmetic. +- **The 5×N exponent** was measured at 11.6/16.2/21.8 against my claimed + 10/15/20 — close enough that the exponent was not where C6's error was. + +--- + +## What the review changed + +| | before | after | +|---|---|---| +| headline finding | *the game is too easy at 5–6 seats* | **withdrawn** — one dumb policy wins 100%, another 0% | +| per-node cost | 112–161 µs | 3.0–4.1 µs (mine) vs 15.6–20.4 (theirs); **unsettled, T04 must benchmark** | +| exhaustive search | ruled out | **available** for bounded endgame; ADR's premise changes | +| the ratio story | three multipliers | 3p and 4p share the ratio and differ by 12.5 points | +| admissibility | claimed | **fails clauses 2 and 3** of a rule one day old | +| explains the maintainer | asserted | contradicted by `lib.rs:2487` | + +**The number was going to ground-game.** T06 exists to report to +GROUND-WP-0005, which is *blocked waiting on a difficulty baseline*. Had +this pass proceeded, they would have been told the game is too easy at +five and six seats and invited to move thresholds — on the strength of one +bot's behaviour, when a second bot of the same sophistication scores zero. + +**That would have been the fifth wrong premise this project has sent +them, and the second stopped by an adversarial review rather than by a +control.** Two for two: both tier-L reviews in this project have caught a +false headline that every gate passed. diff --git a/research/CB-RES-0008-could-we-have-won.md b/research/CB-RES-0008-could-we-have-won.md index 23100d4..307c98d 100644 --- a/research/CB-RES-0008-could-we-have-won.md +++ b/research/CB-RES-0008-could-we-have-won.md @@ -1,7 +1,7 @@ --- id: CB-RES-0008 capability: analysis.witness-and-difficulty -status: draft — awaiting adversarial review (CB-WP-0025 T02) +status: reviewed 2026-08-05 — headline finding WITHDRAWN (C4); numbers corrected (C1) tier: L chaos: d8 = 6 → no override --- @@ -39,14 +39,39 @@ bot win rate over 200 seeds (GR-E01 group success): 4p random 16/200 won = 8.0% mean total 3.2 of 7.0 median margin -4 search cost — legal_commands at every real decision point: - 2p 462 decisions branching mean 4.7 median 5 max 7 161 us/node - 3p 649 decisions branching mean 7.4 median 8 max 10 139 us/node - 4p 870 decisions branching mean 9.1 median 10 max 12 112 us/node + 2p 462 decisions branching mean 4.7 median 5 max 7 3.0 us/call + 3p 649 decisions branching mean 7.4 median 8 max 10 3.5 us/call + 4p 870 decisions branching mean 9.1 median 10 max 12 4.1 us/call ``` -### 1.1 The finding this produced before any solver exists +*(Search-cost row re-measured after C1. The first published figures — +161/139/112 "us/node" — timed two `setup`s, a whole greedy game and a +validate+fold replay. See §1.2.)* -**A greedy bot wins 200 of 200 games at five and six seats.** Not 95%, +### 1.1 WITHDRAWN — the finding this claimed, and why it is not one + +> **Withdrawn 2026-08-05 by the adversarial review (C4), before it left +> the repo.** The section is kept, struck through, because a claim +> retracted silently is how three earlier wrong premises survived +> (ADR-0012 D5). +> +> A `FirstLegal` policy — take `legal[0]`, no heuristic — scores +> **0% at five and six seats** where `GreedyPolicy` scores 100%, and +> **77.5% at two seats** where greedy scores 66%. Two unsophisticated +> agents span the entire range at the same seat count. **A measurement +> that does that is about the policy, not about the game.** +> +> | seats | greedy | firstlegal | +> |---|---:|---:| +> | 2 | 66.0% | **77.5%** | +> | 5 | **100.0%** | **0.0%** | +> | 6 | **100.0%** | **0.0%** | +> +> The arithmetic below is right; the **interpretation** is not, and the +> table also fails GameDesign §1.2 by reporting **sums** where +> GROUND-WP-0004 T02 requires per-priority rows. + +~~**A greedy bot wins 200 of 200 games at five and six seats.**~~ Not 95%, not 99% — every game, with a median margin of **+3** and mean available points of **11.8–12.0 against a threshold of 9**. @@ -64,36 +89,69 @@ table *and* a proportionally lower bar *and* more actions per round to clear it with. Three multipliers all pointing the same direction, which is why the curve is not gentle — it is 66% → 100% across four seat counts. -**This is admissible under GameDesign §1**: the reproduction exists -(`examples/difficulty-baseline.rs`), it has the ruled shape (row-level, no -sums), and it can fail — change a threshold and the numbers move. +~~**This is admissible under GameDesign §1.**~~ **It is not (C3), and the +rule it fails was one day old.** -It also **explains the maintainer's report** — *"I felt it was too easy +- **Clause 2, ruled shape.** 6/9/12 are **sums**, and the `hidden dealt` + column is a *count*. GROUND-WP-0004 T02 requires Surface and each hidden + priority listed separately and explicitly forbids "deal depth N". The + table reproduces the prohibited shape while citing the ruling. +- **Clause 3, can fail.** The harness has **no assertions, no + `--self-test`, and is in no `make` target**. Nothing can turn it red. + Under CB-WP-0022 T05's own `role` distinction it is a `default` artifact + — it prints what the code does — wearing a `counterexample` label. + +**A reproduction that cannot fail is a printout.** T05 must fix the +harness before any figure from it is quoted again. + +~~It also **explains the maintainer's report** — *"I felt it was too easy but then we lost, so who knows"* — without needing a solver. He plays at low seat counts, where 66% is a real game, and had been feeling the 5–6 seat experience from elsewhere in the same session. Both halves of the -sentence are true of different seat counts. +sentence are true of different seat counts.~~ -**T06 must report this to GROUND-WP-0005, which is blocked on exactly -this number.** +**Also withdrawn (C5).** `games/ground/src/lib.rs:2487-2493` records what +actually happened: *"the maintainer played several 3-player games on +2026-08-03 and could not win any of them, because GR-S01 dealt 2/3/4 +Problems worth 3/6/10 against thresholds of 5/7/9."* Three seats, on the +pre-ruling deal, **arithmetically unwinnable at 6 against 7** — nothing to +do with the curve fitted to it here. + +~~**T06 must report this to GROUND-WP-0005.**~~ **It must not.** +GROUND-WP-0005 is blocked waiting on a difficulty baseline, and this is +not one. Sending it would have invited ground-game to move thresholds on +the strength of one bot's behaviour — the **fifth** wrong premise this +project has sent them, and the second stopped by a review rather than by +a control. ### 1.2 What the search-cost numbers rule out -Branching is small — mean 4.7 to 9.1 — but **`legal_commands` costs -112–161 µs per call**, because it constructs candidates and filters them -through full `validate`. That is the price of keeping the rules in one -place (ADR: `legal_commands` builds then validates), and it is the number -that decides this pass. +Branching is small — mean 4.7 to 9.1. -A game is 5 rounds × N seats of decisions. An exhaustive search from -round 1 at 3 seats is roughly `7.4^15` — not a number worth writing down. -**Exhaustive search is out at any seat count**, and this was measured -rather than assumed. +> **CORRECTED (C1). The published figure of 112–161 µs/node was wrong by +> 30–50×**: the timer bracketed two `setup`s, a whole greedy game and a +> validate+fold replay, divided by player-decision count. The tell was in +> the output — it **fell** as branching **rose**, which no per-enumeration +> cost can do. +> +> Re-measured, clock around `legal_commands` only: **3.0 / 3.5 / 4.1 µs** +> at 2/3/4 seats, now rising with branching. **The reviewer measured +> 15.6–20.4 µs by a different isolation and we have not settled which is +> right** — T04 must benchmark it properly (`criterion` is already a +> dev-dependency) rather than adopt either. -What is affordable, at ~140 µs/node: a bounded search of ~10⁴–10⁵ nodes -costs **1.4–14 seconds**. That is the budget the ADR has to design inside, -and it is the difference between a feature that answers while the player -is still looking at the page and one that does not. +~~**Exhaustive search is out at any seat count.**~~ **False (C6).** The +reviewer measured ~16 minutes for a full 2p game and **~3 seconds over the +last two rounds at 3p**. With C1's correction the affordable budget is +**~10⁵–10⁶ nodes**, and bounded exhaustive search over the endgame is +inside it. + +**This changes T03's starting point.** The ADR cannot open with +*"exhaustive is impossible, therefore determinized sampling"* — the +premise is false, and the alternative carries strategy fusion that +exhaustive search does not. Neither the survey nor the review considered +transposition or the co-operative single-agent framing, which cut the +exponent further. ## 2. Prior art: determinized search, and the failure it is famous for diff --git a/workplans/CB-WP-0025-could-we-have-won.md b/workplans/CB-WP-0025-could-we-have-won.md index f46216c..23b4739 100644 --- a/workplans/CB-WP-0025-could-we-have-won.md +++ b/workplans/CB-WP-0025-could-we-have-won.md @@ -160,7 +160,7 @@ them. ```task id: CB-WP-0025-T02 -status: todo +status: done priority: high state_hub_task_id: "bdd9fc97-4706-4d8e-b286-91ff53339680" ``` @@ -184,6 +184,53 @@ Tier L requires it. Exactly one round: challenge, then response, trail in *produces* them, but the reviewer should press whether that is a distinction worth a separate capability. +**Done 2026-08-05.** Trail: +[challenge](../history/260805-could-we-have-won-challenge.md), +[response](../history/260805-could-we-have-won-response.md). +Separate agent, as in CB-WP-0022. + +**Six of seven conceded, and the survey's headline finding is withdrawn — +not softened.** + +**C4 is the one that kills it, and the reviewer ranked it fourth.** +Measured: a `FirstLegal` policy — take `legal[0]`, no heuristic — scores +**0% at five and six seats** where greedy scores 100%, and **77.5% at two +seats** where greedy scores 66%. **Two unsophisticated agents span the +entire range at the same seat count**, so *"the game is too easy at 5–6 +seats"* is a statement about `GreedyPolicy`, not about GROUND. Their +offered rescue — greedy hits the 12-point ceiling every time, so it is a +rules claim — dies on the same data. + +**C1: the per-node cost was wrong by 30–50×.** The timer bracketed two +`setup`s, a whole greedy game and a validate+fold replay, then divided by +player decisions. **The tell was in my own published output**: the figure +*fell* as branching *rose*, which no per-enumeration cost can do. +Re-measured at **3.0–4.1 µs**; the reviewer got 15.6–20.4 by a different +isolation and **that discrepancy is unsettled** — T04 benchmarks it. + +**C6: exhaustive search is not out**, which changes T03's premise. With +C1's correction the budget is ~10⁵–10⁶ nodes and bounded endgame search +fits — so the ADR cannot open with *"exhaustive is impossible, therefore +determinized sampling"*, especially since sampling carries strategy fusion +that exhaustive search does not. + +**C3: the finding failed the admissibility rule this project wrote nine +hours earlier** — sums where GROUND-WP-0004 T02 requires per-priority +rows, and a harness with no assertions, no `--self-test` and no `make` +target, so nothing can turn it red. A `default` artifact wearing a +`counterexample` label. + +**C2: the ratio explains nothing** — 3p and 4p share deal, threshold and +ratio, and differ by 12.5 points of win rate. **C5: the "explains the +maintainer's report" claim is contradicted by `lib.rs:2487`**, which +records his losses as 3-player on the pre-ruling deal, unwinnable at 6 +against 7. + +**T06 was pointed at GROUND-WP-0005, which is blocked on this number.** +Sending it would have invited threshold changes on one bot's behaviour — +the fifth wrong premise to reach ground-game. **Both tier-L reviews in +this project have now caught a false headline that every gate passed.** + ## Task: decide ```task