//! AM-6/AM-7 benchmarks (GameKernel §4): the real GROUND aggregate under //! the CB-RES-0001 synthetic workload (3-player commit/reveal rounds). //! //! The baseline is boardgame.io, recorded in //! research/CB-RES-0001-harness/boardgame-io/results-260731.json. That //! harness measured moves/second *while history grew*, and its headline //! finding was that throughput halved as history doubled. AM-7 exists //! because of that finding, so the same workload sizes are run here: //! what matters is the shape of the curve, not only the peak number. use cb_events::state_hash_hex; use cb_game_runtime::{ScenarioFile, ScenarioGame, Setup}; use cb_kernel::{Actor, Aggregate, PlayerId}; use criterion::{criterion_group, criterion_main, BatchSize, Criterion, Throughput}; use games_ground::{Action, GroundCommand, GroundMode, GroundState}; use std::collections::BTreeMap; fn setup(seed: u64) -> GroundState { GroundState::setup( &Setup { players: 3, preset: "standard-3p".to_string(), patch: BTreeMap::new(), }, seed, ) .expect("standard-3p setup") } fn apply(state: &mut GroundState, actor: Actor, command: &GroundCommand) -> usize { match state.validate(actor, command) { Ok(events) => { for event in &events { state.fold(event); } events.len() } Err(_) => 0, } } /// K18: the benchmark workload, **loaded from the scenario format**. /// /// Until CB-WP-0006 T07 this function hardcoded the command sequence in /// Rust and never touched `ScenarioFile`, so K18 — "Criterion benches /// driving the same scenario format at scale" — was false. Embedding the /// file at compile time keeps the bench self-contained while making the /// workload *data*: editing `benchmarks/synthetic-3p.yaml` changes what /// AM-6 and AM-7 measure, and `bench_shape` in the aggregate crate breaks /// if that changes the round's shape. const WORKLOAD_YAML: &str = include_str!("../../../benchmarks/synthetic-3p.yaml"); fn workload() -> Vec<(Actor, GroundCommand)> { let sc = ScenarioFile::from_yaml(WORKLOAD_YAML).expect("bench workload parses"); // Positive control: an empty or mis-parsed workload would benchmark // nothing while still reporting a rate. assert!( !sc.commands.is_empty(), "K18: the bench workload has no commands" ); sc.commands .iter() .map(|step| GroundState::parse_command(step).expect("bench command parses")) .collect() } /// One full round for three players, replayed from the scenario workload. /// Returns the number of events applied. fn play_round(state: &mut GroundState, round: &[(Actor, GroundCommand)]) -> usize { let mut applied = 0; for (actor, command) in round { applied += apply(state, *actor, command); } applied } fn run_rounds(rounds: usize) -> usize { let mut applied = 0; let mut state = setup(42); let script = workload(); for round in 0..rounds { if state.outcome.is_some() { state = setup(42 + round as u64); } let produced = play_round(&mut state, &script); // A workload whose commands get rejected still "runs", but it // measures nothing. An earlier version of this bench stalled on // the GR-R03 stress gate and reported throughput for rounds that // never happened, so refuse to measure that. assert!( produced == EVENTS_PER_ROUND || produced == FINAL_ROUND_EVENTS, "round {round} produced {produced} events, expected {EVENTS_PER_ROUND} \ (or {FINAL_ROUND_EVENTS} on a game's last round)" ); applied += produced; } applied } /// Pinned by the `bench_shape` test in the aggregate crate. const EVENTS_PER_ROUND: usize = 13; /// GR-R09: a game's fifth round emits GameEnded instead of RoundEnded /// plus StepAdvanced, so it is one event shorter. const FINAL_ROUND_EVENTS: usize = 12; /// Mean events per round over a 5-round game, scaled by 5 to stay in /// integers: (4 x 13 + 12) = 64. const EVENTS_PER_5_ROUNDS: usize = 64; /// Play one round, appending every applied event to `log`. fn record_round(state: &mut GroundState, log: &mut Vec) { let mut run = |state: &mut GroundState, actor: Actor, cmd: &GroundCommand| { if let Ok(produced) = state.validate(actor, cmd) { for event in &produced { state.fold(event); log.push(event.clone()); } } }; for (seat, action, target) in [ (0u8, Action::Attack, Some(PlayerId(1))), (2, Action::Support, Some(PlayerId(1))), (1, Action::Ground, None), ] { run( state, Actor::Player(PlayerId(seat)), &GroundCommand::SelectAction { action, target, problem: None, }, ); } run(state, Actor::System, &GroundCommand::Reveal); run( state, Actor::Player(PlayerId(1)), &GroundCommand::ChooseGroundMode { mode: GroundMode::Gr, choice: None, }, ); run(state, Actor::System, &GroundCommand::Resolve); run(state, Actor::System, &GroundCommand::EndRound); } fn bench_synthetic(c: &mut Criterion) { // Events per round is fixed by the workload, so throughput can be // reported in events/second — the AM-6 unit. assert_eq!(play_round(&mut setup(1), &workload()), EVENTS_PER_ROUND); let mut group = c.benchmark_group("synthetic-ground-3p"); // AM-6 headline throughput and AM-7 scaling, at the sizes the // boardgame.io harness used so the curves line up. for &rounds in &[5_000usize, 10_000, 20_000, 40_000, 100_000] { group.throughput(Throughput::Elements( (rounds * EVENTS_PER_5_ROUNDS / 5) as u64, )); group.bench_function(format!("rounds-{rounds}"), |b| { b.iter_batched(|| rounds, run_rounds, BatchSize::SmallInput) }); } group.finish(); // AM-7 replay, and the honest analogue of the boardgame.io finding: // a *single* growing event log folded back into state. The round // benches above restart the game every 5 rounds (GR-R09), so their // flat curve is partly by construction — this one is not, because // the log here grows without bound. let mut replay = c.benchmark_group("replay-ground-3p"); for &target_events in &[10_000usize, 100_000] { // Build one log by playing real rounds, then measure folding it // back. Must use the full command sequence: a shortened one // stalls, because Reveal needs every seat's selection and // EndRound is gated on Resolve. let mut log = Vec::with_capacity(target_events); let mut source = setup(42); let mut games = 0u64; while log.len() < target_events { if source.outcome.is_some() { games += 1; source = setup(42 + games); } let before = log.len(); record_round(&mut source, &mut log); // Positive control: a round that yields nothing means the // workload stalled, and the loop above would spin forever. assert!( log.len() > before, "replay workload stalled: a round produced no events" ); } replay.throughput(Throughput::Elements(log.len() as u64)); replay.bench_function(format!("fold-{target_events}-events"), |b| { b.iter(|| { let mut state = setup(42); for event in &log { state.fold(event); } state_hash_hex(&state) }) }); } replay.finish(); // AM-7: hashing the full aggregate, the per-round determinism cost. let mut hashing = c.benchmark_group("state-hash-ground-3p"); hashing.throughput(Throughput::Elements(1)); hashing.bench_function("hash-one-state", |b| { let state = setup(42); b.iter(|| state_hash_hex(&state)) }); hashing.finish(); } criterion_group!(benches, bench_synthetic); criterion_main!(benches);