diff --git a/games/ground/src/bot.rs b/games/ground/src/bot.rs index 3fd8ec0..b160058 100644 --- a/games/ground/src/bot.rs +++ b/games/ground/src/bot.rs @@ -820,4 +820,74 @@ mod tests { ); } } + /// GR-O01 says 2–6 players. Every scenario in the corpus was + /// 3-player until CB-WP-0008 T03, so the range was stated and tested + /// at one point. This plays all five counts with both policies. + #[test] + fn every_seat_count_in_gr_o01_plays_to_the_end() { + for players in 2..=6u8 { + for kind in ["random", "greedy"] { + let game = play(setup(players, 42), &mut policies(kind, players, 42)) + .unwrap_or_else(|e| panic!("{players}p {kind}: {e}")); + let outcome = game + .state + .outcome + .as_ref() + .unwrap_or_else(|| panic!("{players}p {kind}: no outcome")); + assert_eq!(game.state.round, 5, "{players}p {kind}: GR-R09"); + assert_eq!( + outcome.personal.len(), + usize::from(players), + "{players}p {kind}: every seat must be scored" + ); + // K8 at each boundary, not only at three seats. + let again = + play(setup(players, 42), &mut policies(kind, players, 42)).expect("second run"); + assert_eq!( + cb_events::state_hash_hex(&game.state), + cb_events::state_hash_hex(&again.state), + "{players}p {kind}: same seed must reproduce" + ); + } + } + } + + /// **A recorded finding, not a desired property.** With the standard + /// preset's placeholder Problem values (value = priority), the total + /// a game can possibly reach is below GR-E01's threshold at 2, 3 and + /// 4 players — group success is unreachable regardless of play. Only + /// 5–6p can clear its 9. + /// + /// This test pins the arithmetic so the gap cannot close silently. + /// **It is expected to fail** when Problem values become scenario + /// data (GR-S01 calls the current fixture a stand-in); the failure is + /// the signal to delete it, not to re-tune it. + #[test] + fn the_standard_preset_cannot_reach_the_threshold_below_five_seats() { + let mut report = Vec::new(); + for players in 2..=6u8 { + let initial = setup(players, 42); + let best: u32 = initial.problems.values().map(|p| u32::from(p.value)).sum(); + let threshold = play(initial, &mut policies("greedy", players, 42)) + .expect("game") + .state + .outcome + .expect("outcome") + .threshold; + report.push(format!("{players}p best {best} vs threshold {threshold}")); + if players < 5 { + assert!( + best < threshold, + "{players}p: best {best} now reaches threshold {threshold} — \ + the fixture changed, delete this test" + ); + } else { + assert!( + best >= threshold, + "{players}p: best {best} cannot reach threshold {threshold}" + ); + } + } + println!("GR-E01 reachability: {}", report.join(", ")); + } } diff --git a/history/260801-cb-wp-0008-log.md b/history/260801-cb-wp-0008-log.md index 609c34d..a3fac7b 100644 --- a/history/260801-cb-wp-0008-log.md +++ b/history/260801-cb-wp-0008-log.md @@ -100,3 +100,39 @@ This is the same class as the AM-2 `expect` that matched passing output — an assertion that cannot distinguish the two worlds. The remedy that worked both times was **counting what the harness examined**, not strengthening the predicate. + +## T03 — the 2–6 player range + +Every scenario in the corpus was 3-player. Now: `gr-o01-two-player`, +`gr-o01-six-player`, `gr-e01-threshold-unreachable-2p`, plus all-bot +games at every count under both policies, each reproducing at the same +seed (K8 at the boundaries, not only in the middle), and the CLI +transcript test run at 2p, 3p and 6p. + +**Nothing broke.** The rules are seat-count-generic and the range works. +What the boundaries exposed is arithmetic: + +| seats | Problems | best possible total | GR-E01 threshold | group success | +|---|---|---|---|---| +| 2 | 2 | 3 | 5 | **unreachable** | +| 3 | 3 | 6 | 7 | **unreachable** | +| 4 | 3 | 6 | 7 | **unreachable** | +| 5 | 4 | 10 | 9 | reachable | +| 6 | 4 | 10 | 9 | reachable | + +With the standard preset's placeholder values (value = priority), *no +play at all* can clear the threshold below five seats. GR-S01 calls the +fixture a stand-in for scenario Problem data, so this is evidence the +stand-in is not neutral — not that GR-E01 is wrong. It is pinned three +ways: a scenario that passes on the fact, a test that asserts the +arithmetic, and a `provisional` marker with `ground-game` as owner so it +ages in `make coverage` (now 6 provisional defaults). + +The test carries its own delete-by condition: **it is expected to fail** +when Problem values become scenario data, and that failure is the signal +to delete it rather than re-tune it. + +Two smaller notes, recorded and not acted on: at 2 players GR-L01's +second relation slot can never be used (there is only one possible +partner), and a human at P1 is always prompted before any other seat has +selected, so the seat order makes P1 a weaker test position than P3. diff --git a/scenarios/ground/gr-e01-threshold-unreachable-2p.yaml b/scenarios/ground/gr-e01-threshold-unreachable-2p.yaml new file mode 100644 index 0000000..1a7ad5d --- /dev/null +++ b/scenarios/ground/gr-e01-threshold-unreachable-2p.yaml @@ -0,0 +1,53 @@ +scenario: ground/gr-e01-threshold-unreachable-2p +description: > + GR-E01's threshold against the standard preset's Problem values, at the + 2-player boundary. Both Problems are claimed — the best case available + — and the total is 3 against a threshold of 5. With the placeholder + fixture (value = priority) group success is unreachable at 2, 3 and 4 + players; only 5–6p can reach its 9. Recorded here so the gap has a + failing-in-fact scenario rather than a paragraph, and marked provisional + because the fixture is explicitly a stand-in for scenario Problem data. +covers: [GR-E01, GR-E02, GR-O01] +provisional: true +provisional_owner: ground-game +provisional_raised: 2026-08-01 +seed: 42 +setup: + players: 2 + preset: standard-2p + patch: + "round": 5 + "mode": SharedGround + "problems.1.claimed_by": 0 + "problems.2.claimed_by": 1 + "problems.2.face_up": true +commands: + - actor: P1 + cmd: select_action + args: { action: GROUND } + - actor: P2 + cmd: select_action + args: { action: GROUND } + - actor: SYSTEM + cmd: reveal + - actor: P1 + cmd: choose_ground_mode + args: { mode: GR } + - actor: P2 + cmd: choose_ground_mode + args: { mode: GR } + - actor: SYSTEM + cmd: resolve + - actor: SYSTEM + cmd: end_round +expect: + events: + - kind: GameEnded + state: + # 1 + 2 = 3, the maximum any 2-player game of this preset can score. + "outcome.total": 3 + "outcome.threshold": 5 + "outcome.group_success": false + "round": 5 + "step": End + rejects: [] diff --git a/scenarios/ground/gr-o01-six-player.yaml b/scenarios/ground/gr-o01-six-player.yaml new file mode 100644 index 0000000..f39f3f2 --- /dev/null +++ b/scenarios/ground/gr-o01-six-player.yaml @@ -0,0 +1,51 @@ +scenario: ground/gr-o01-six-player +description: > + GR-O01's upper boundary, the counterpart to gr-o01-two-player. Six + seats exercise the 5–6p Problem set (GR-S01, four priorities) and the + longest resolution order (GR-R07, Lead first then clockwise with a + wrap). +covers: [GR-O01, GR-S01, GR-R07] +seed: 42 +setup: + players: 6 + preset: standard-6p +commands: + - actor: P1 + cmd: select_action + args: { action: INVESTIGATE, problem: 2 } + - actor: P2 + cmd: select_action + args: { action: INVESTIGATE, problem: 3 } + - actor: P3 + cmd: select_action + args: { action: INVESTIGATE, problem: 4 } + - actor: P4 + cmd: select_action + args: { action: SUPPORT, target: P5 } + - actor: P5 + cmd: select_action + args: { action: GROUND } + - actor: P6 + cmd: select_action + args: { action: ATTACK, target: P1 } + - actor: SYSTEM + cmd: reveal + - actor: P5 + cmd: choose_ground_mode + args: { mode: GR } + - actor: P5 + cmd: respond_to_support + args: { response: accept_bond } + - actor: SYSTEM + cmd: resolve + - actor: SYSTEM + cmd: end_round +expect: + events: + - kind: ProblemRevealed + - kind: RoundEnded + state: + # GR-S01: four Problems at six players. + "problems.4.face_up": true + "round": 2 + rejects: [] diff --git a/scenarios/ground/gr-o01-two-player.yaml b/scenarios/ground/gr-o01-two-player.yaml new file mode 100644 index 0000000..ebf9a34 --- /dev/null +++ b/scenarios/ground/gr-o01-two-player.yaml @@ -0,0 +1,40 @@ +scenario: ground/gr-o01-two-player +description: > + GR-O01's lower boundary. Every scenario in this corpus before + CB-WP-0008 T03 was 3-player, so a rule stated for a range (2–6) was + tested at one point. A full round at two seats: setup deals the 2p + Problem set (GR-S01), both seats select, and resolution runs with the + smallest possible seat order. +covers: [GR-O01, GR-S01, GR-R01, GR-R07] +seed: 42 +setup: + players: 2 + preset: standard-2p +commands: + - actor: P1 + cmd: select_action + args: { action: INVESTIGATE, problem: 2 } + - actor: P2 + cmd: select_action + args: { action: SUPPORT, target: P1 } + - actor: SYSTEM + cmd: reveal + - actor: P1 + cmd: respond_to_support + args: { response: accept_bond } + - actor: SYSTEM + cmd: resolve + - actor: SYSTEM + cmd: end_round +expect: + events: + # GR-R06 order: Support (step 2) resolves before INVESTIGATE (step 4). + - kind: RelationFormed + - kind: ProblemRevealed + - kind: RoundEnded + state: + # GR-S01: two Problems at two players, not three. + "problems.2.face_up": true + "round": 2 + "step": Select + rejects: [] diff --git a/tools/cb-play/src/main.rs b/tools/cb-play/src/main.rs index 9426824..eabf83f 100644 --- a/tools/cb-play/src/main.rs +++ b/tools/cb-play/src/main.rs @@ -142,37 +142,42 @@ mod tests { list.iter().map(|s| s.to_string()).collect() } - /// A scripted transcript plays a full 3-player game to `GameEnded`. - /// "0" always takes the first legal command, which is a real player - /// input and needs no knowledge of the board. + /// A scripted transcript plays a full game to `GameEnded`. "0" + /// always takes the first legal command, which is a real player input + /// and needs no knowledge of the board. + /// + /// Run at **both GR-O01 boundaries and the middle** (T03): the CLI is + /// where a seat-count assumption would show up as a prompt nobody can + /// answer. #[test] fn a_scripted_transcript_plays_a_full_game() { - let config = Config { - seed: 42, - players: 3, - human_seats: vec![0], - bot: "greedy".into(), - replay_dir: None, - record: None, - }; - let script = "0\n".repeat(200); - let mut out: Vec = Vec::new(); - let summary = table::play(&config, script.as_bytes(), &mut out) - .unwrap_or_else(|e| panic!("scripted game failed: {e}")); + for players in [2u8, 3, 6] { + let config = Config { + seed: 42, + players, + human_seats: vec![0], + bot: "greedy".into(), + replay_dir: None, + record: None, + }; + let script = "0\n".repeat(400); + let mut out: Vec = Vec::new(); + let summary = table::play(&config, script.as_bytes(), &mut out) + .unwrap_or_else(|e| panic!("{players}p scripted game failed: {e}")); - assert_eq!(summary.rounds, 5, "GR-R09 runs five rounds"); - let text = String::from_utf8(out).expect("utf8"); - assert!(text.contains("OUTCOME"), "the game must report an outcome"); - assert!( - text.contains("you are P1"), - "the human seat must be prompted" - ); + assert_eq!(summary.rounds, 5, "{players}p: GR-R09 runs five rounds"); + let text = String::from_utf8(out).expect("utf8"); + assert!(text.contains("OUTCOME"), "{players}p: no outcome reported"); + assert!(text.contains("you are P1"), "{players}p: no prompt"); - // The transcript replays identically — through the scenario - // runner, not through a second call to the same driver. - match run::(&summary.scenario) { - RunOutcome::Passed { .. } => {} - RunOutcome::Failed { reason, .. } => panic!("replay failed: {reason}"), + // The transcript replays identically — through the scenario + // runner, not through a second call to the same driver. + match run::(&summary.scenario) { + RunOutcome::Passed { .. } => {} + RunOutcome::Failed { reason, .. } => { + panic!("{players}p replay failed: {reason}") + } + } } } diff --git a/workplans/CB-WP-0008-ship-stage-0.md b/workplans/CB-WP-0008-ship-stage-0.md index 056b97e..58495ef 100644 --- a/workplans/CB-WP-0008-ship-stage-0.md +++ b/workplans/CB-WP-0008-ship-stage-0.md @@ -97,7 +97,7 @@ records as a scenario and as a `.cbreplay` bundle. Notes in ```task id: CB-WP-0008-T03 -status: todo +status: done priority: medium state_hub_task_id: "438d21b7-98f3-43c0-b17f-6ae32bf176bd" ``` @@ -112,6 +112,12 @@ all-bot game at each, and report what breaks. Finding that 2p or 6p does *not* work is a legitimate and likely outcome — record it rather than quietly narrowing GR-O01. +**Done 2026-08-01.** All five counts play to `GameEnded` under both +policies and reproduce at the same seed; three new scenarios; and the +finding is arithmetic, not a crash — GR-E01's threshold is **unreachable +at 2, 3 and 4 players** with the placeholder Problem values. Table in +`history/260801-cb-wp-0008-log.md`. + ## Task: evidence and retrospective ```task