From 82fdb851738d49a2c4c2a3d2b59a1414b3e812ea Mon Sep 17 00:00:00 2001 From: tegwick Date: Sun, 2 Aug 2026 02:16:51 +0200 Subject: [PATCH] CB-WP-0010-T02: GR-E03 gets a scenario MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit COMMON PROBLEM was the only scoring mode with no scenario — implemented since CB-WP-0001, referenced by nothing, and would not have failed if deleted. Five players, where the threshold is reachable at all: every Problem claimed for a total of 10 against 9, and the winner is decided by Blame rather than by claimed value — P4 claimed the highest Problem and loses to P3 because two Blame tokens sit in front of them. No defect was found on the first execution of that path, which is worth saying plainly rather than implying the scenario was hard-won. The assertions were mutation-checked three ways instead of trusted, because a dot-path expectation that silently fails to resolve would report PASS: a wrong personal score, a wrong mastery, and removing the Blame from the fixture each turn it red. Co-Authored-By: Claude Opus 5 --- facts.toml | 6 +- scenarios/ground/gr-e03-common-problem.yaml | 81 +++++++++++++++++++++ workplans/CB-WP-0010-consolidation.md | 8 +- 3 files changed, 91 insertions(+), 4 deletions(-) create mode 100644 scenarios/ground/gr-e03-common-problem.yaml diff --git a/facts.toml b/facts.toml index fd99502..06a215c 100644 --- a/facts.toml +++ b/facts.toml @@ -6,7 +6,7 @@ # `make facts-check` fails if this file disagrees with the # instruments, or if a tagged artifact disagrees with this file. -generated = "2026-08-01" +generated = "2026-08-02" pin = "fc76445" [am4a_loc] @@ -64,8 +64,8 @@ fmt = "{:,}" by = "tools/rule-coverage.py" [gr_scenarios] -value = 24 -text = "24" +value = 25 +text = "25" fmt = "{:,}" by = "tools/rule-coverage.py" diff --git a/scenarios/ground/gr-e03-common-problem.yaml b/scenarios/ground/gr-e03-common-problem.yaml new file mode 100644 index 0000000..ddf7a08 --- /dev/null +++ b/scenarios/ground/gr-e03-common-problem.yaml @@ -0,0 +1,81 @@ +scenario: ground/gr-e03-common-problem +description: > + COMMON PROBLEM scoring (GR-E03), the only scoring mode with no scenario + until CB-WP-0010 T02 — implemented since CB-WP-0001 and referenced by + nothing. Five players, where the threshold is actually reachable (10 + available against 9). The group qualifies, personal scores are claimed + value −1 per Blame held, and the Blame is what decides the winner: P4 + claimed the highest-value Problem and still loses to P3 because two + Blame tokens sit in front of them. +covers: [GR-E03, GR-E01, GR-R09, GR-T02] +seed: 42 +setup: + players: 5 + preset: standard-5p + patch: + "round": 5 + "mode": CommonProblem + # Every Problem claimed: 1+2+3+4 = 10, over the 5-6p threshold of 9. + "problems.1.claimed_by": 0 + "problems.2.claimed_by": 1 + "problems.3.claimed_by": 2 + "problems.4.claimed_by": 3 + "problems.2.face_up": true + "problems.3.face_up": true + "problems.4.face_up": true + # GR-T02: two Blame tokens in front of P4, −1 each. + "players.3.blame_from": [0, 1] +commands: + - actor: P1 + cmd: select_action + args: { action: GROUND } + - actor: P2 + cmd: select_action + args: { action: GROUND } + - actor: P3 + cmd: select_action + args: { action: GROUND } + - actor: P4 + cmd: select_action + args: { action: GROUND } + - actor: P5 + cmd: select_action + args: { action: GROUND } + - actor: SYSTEM + cmd: reveal + - actor: P1 + cmd: choose_ground_mode + args: { mode: GR } + - actor: P2 + cmd: choose_ground_mode + args: { mode: GR } + - actor: P3 + cmd: choose_ground_mode + args: { mode: GR } + - actor: P4 + cmd: choose_ground_mode + args: { mode: GR } + - actor: P5 + cmd: choose_ground_mode + args: { mode: GR } + - actor: SYSTEM + cmd: resolve + - actor: SYSTEM + cmd: end_round +expect: + events: + - kind: GameEnded + state: + "outcome.total": 10 + "outcome.threshold": 9 + "outcome.group_success": true + # GR-E03: claimed value −1 per Blame held. + "outcome.personal.3": 2 + "outcome.personal.2": 3 + # P4 claimed 4 and still loses: the Blame is load-bearing here. + "outcome.winners": [2] + # GR-E03 has no Mastery rating; that is GR-E02's. + "outcome.mastery": null + "round": 5 + "step": End + rejects: [] diff --git a/workplans/CB-WP-0010-consolidation.md b/workplans/CB-WP-0010-consolidation.md index 3b31d00..02c8a79 100644 --- a/workplans/CB-WP-0010-consolidation.md +++ b/workplans/CB-WP-0010-consolidation.md @@ -46,7 +46,7 @@ is what it has been doing for two passes. ```task id: CB-WP-0010-T02 -status: todo +status: done priority: high state_hub_task_id: "89355b0a-9fc7-4d69-a27e-d3c94ff0730f" ``` @@ -63,6 +63,12 @@ is actually reachable (10 available against 9). **Finding a defect here is the likely outcome and is the point** — this is the first execution of that code path. +**Done 2026-08-02.** `gr-e03-common-problem.yaml` passes, and **no +defect was found** — GR-E03 scored correctly on its first execution. +The assertions were mutation-checked three ways rather than trusted: +a wrong personal score, a wrong `mastery`, and removing the Blame from +the fixture each turn it red. Every scoring mode now has a scenario. + ## Task: record CommitWindow's second failed use ```task