From 0f8559f442024885fb47c39b1f21f4fb799aa7b6 Mon Sep 17 00:00:00 2001 From: tegwick Date: Mon, 28 Sep 2026 14:03:29 +0200 Subject: [PATCH] Require boolean judgments and intact actor isolation guards Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4 --- docs/TestDriverGeneralisationReview.md | 21 +++++ research/decisions/README.md | 1 + src/testdriver/oracles.py | 5 + src/testdriver/runner.py | 24 ++++- tests/test_guard_integrity.py | 94 +++++++++++++++++++ workplans/TD-WP-0003-generalise-and-settle.md | 13 +++ 6 files changed, 156 insertions(+), 2 deletions(-) create mode 100644 tests/test_guard_integrity.py diff --git a/docs/TestDriverGeneralisationReview.md b/docs/TestDriverGeneralisationReview.md index 0bceb73..1ce0ba3 100644 --- a/docs/TestDriverGeneralisationReview.md +++ b/docs/TestDriverGeneralisationReview.md @@ -271,3 +271,24 @@ through direct and composite dispatch, driver reuse and invalid frozen lookup. Decision: `e8fabf6e-3e5e-424e-9a60-152bf3dec641`. Validation: the complete suite passed **343 tests** (153.67 seconds). + +## Predicate and isolation guard integrity (2026-09-28) + +Oracle predicates must return actual booleans. None, numbers, strings, +containers and objects now produce INCONCLUSIVE; their truth-value methods are +never invoked. Existing True/False results retain PASS/FAIL semantics. Predicate +results are not copied into the diagnostic, avoiding disclosure of their values. + +At each existing isolation boundary, the runner now checks for shared memory +stores, missing or changed stored canaries, and invalid or duplicate actor +canaries. Removing markers can no longer make shared actor memory appear clean. +These findings use the existing abort/acceptance/crystallization gates and record +actor identities without marker or private memory values. This remains a +boundary diagnostic, not a sandbox against arbitrary or transient memory access. + +Twenty regressions cover claims and invariants, non-coercible objects, real +boolean controls, invalid-result admission and crystallization, plus broken +isolation guards before and during runs. Sixteen failed before implementation; +the focused suite passed 69 tests afterward. Decision: `2ecac1c5-5254-4ce9-b092-b47c3e1283a9`. + +Full-suite validation: **363 tests passed** in 153.10 seconds. diff --git a/research/decisions/README.md b/research/decisions/README.md index c6d9050..5bd8e2d 100644 --- a/research/decisions/README.md +++ b/research/decisions/README.md @@ -11,3 +11,4 @@ file is a pointer table, not a second source of truth. | TD-WP-0003 T08 safety follow-up | Run-input manifests and incomplete-run handling | `afb38ade-893b-48a2-b1b3-c55e49cc40ee` | `docs/TestDriverGeneralisationReview.md` | | TD-WP-0003 T08 admission follow-up | Qualified crystallization, passing baselines and intent revisions | `88e51e10-cd32-44d2-a2d6-055675b5ce04` | `docs/TestDriverGeneralisationReview.md` | | TD-WP-0003 T08 isolation/replay follow-up | Isolation acceptance gate and step-bound frozen replay | `e8fabf6e-3e5e-424e-9a60-152bf3dec641` | `docs/TestDriverGeneralisationReview.md` | +| TD-WP-0003 T08 guard-integrity follow-up | Strict boolean judgments and intact isolation guards | `2ecac1c5-5254-4ce9-b092-b47c3e1283a9` | `docs/TestDriverGeneralisationReview.md` | diff --git a/src/testdriver/oracles.py b/src/testdriver/oracles.py index 17f1c20..8fa7860 100644 --- a/src/testdriver/oracles.py +++ b/src/testdriver/oracles.py @@ -83,6 +83,11 @@ class Oracle: step_id, {"reason": f"predicate raised {type(exc).__name__}: {exc}"}, ) + if type(satisfied) is not bool: + return Judgment( + assertion.id, assertion.text, Verdict.INCONCLUSIVE, step_id, + {"reason": "predicate did not return a boolean"}, + ) return Judgment( assertion.id, assertion.text, diff --git a/src/testdriver/runner.py b/src/testdriver/runner.py index 28a1f1b..3fdb6d1 100644 --- a/src/testdriver/runner.py +++ b/src/testdriver/runner.py @@ -57,12 +57,32 @@ class Runner: # -- independence guards --------------------------------------------- def _isolation_violations(self) -> list[str]: - """Does any actor hold another's canary? + """Check separate stores, intact unique markers and cross-actor leaks. Run on every scenario, not only on ones written to test isolation. """ - canaries = {actor.canary: actor.id for actor in self._world.cast} + canaries: dict[str, str] = {} + stores: dict[int, str] = {} violations: list[str] = [] + for actor in self._world.cast: + store = id(actor._memory) + if store in stores: + violations.append( + f"actor {actor.id!r} shares a memory store with {stores[store]!r}" + ) + stores[store] = actor.id + if not isinstance(actor.canary, str) or not actor.canary: + violations.append(f"actor {actor.id!r} has an invalid private marker") + continue + if actor.canary in canaries: + violations.append( + f"actor {actor.id!r} has a duplicate private marker" + ) + canaries[actor.canary] = actor.id + if actor.recall("__canary__") != actor.canary: + violations.append( + f"actor {actor.id!r} has a missing or changed private marker" + ) for actor in self._world.cast: for key in actor.known_keys(): value = actor.recall(key) diff --git a/tests/test_guard_integrity.py b/tests/test_guard_integrity.py new file mode 100644 index 0000000..2e0df96 --- /dev/null +++ b/tests/test_guard_integrity.py @@ -0,0 +1,94 @@ +"""Reject invalid predicate results and broken actor isolation instrumentation.""" +from dataclasses import replace +import json + +import pytest + +from scenarios.alice_bob_carol import build +from testdriver import Runner, Verdict +from testdriver.classification import classify +from testdriver.crystallization import assess_stability +from testdriver.oracles import Oracle + + +class CannotCoerce: + def __bool__(self): + raise AssertionError('oracle must not coerce predicate results') + + +@pytest.mark.parametrize('value', [None, 'unknown', {}, [], 0, 1, CannotCoerce()]) +def test_non_boolean_predicate_results_are_inconclusive(value): + _, _, _, asset, _ = build() + for assertion in (*asset.scenario.use_case.claims, *asset.scenario.use_case.invariants): + assertion = replace(assertion, predicate=lambda snapshot: value) + judgment = Oracle().judge(assertion, {'observed': True}, 'step') + assert judgment.verdict is Verdict.INCONCLUSIVE + assert 'boolean' in judgment.detail['reason'] + + +@pytest.mark.parametrize('value,verdict', [(True, Verdict.PASS), (False, Verdict.FAIL)]) +def test_boolean_predicate_results_keep_their_meaning(value, verdict): + _, _, _, asset, _ = build() + assertion = replace(asset.scenario.use_case.claims[0], predicate=lambda snapshot: value) + assert Oracle().judge(assertion, {'observed': True}, 'step').verdict is verdict + + +@pytest.mark.parametrize('value', [None, 'unknown', 1]) +def test_invalid_predicates_cannot_authorize_acceptance_or_crystallization(value): + packs = [] + for _ in range(3): + world, driver, observer, asset, oracle = build() + case = asset.scenario.use_case + asset.scenario = replace(asset.scenario, use_case=replace(case, claims=tuple( + replace(c, predicate=lambda snapshot: value) for c in case.claims))) + result = Runner(world, driver, observer, oracle).run(asset) + assert result.verdict is Verdict.INCONCLUSIVE + packs.append(json.loads(result.evidence.to_json())) + assert not classify(packs[0], packs[1]).safe_to_accept + assert not assess_stability(packs).stable + + +@pytest.mark.parametrize('damage', ['shared', 'missing', 'changed', 'duplicate']) +@pytest.mark.parametrize('during_run', [False, True]) +def test_corrupt_isolation_guards_stop_execution(damage, during_run): + packs = [] + for _ in range(3): + world, driver, observer, asset, oracle = build() + calls = [] + + def corrupt(): + alice, bob = world.cast['alice'], world.cast['bob'] + if damage == 'shared': + alice._memory = bob._memory = {} + alice.remember('private', 'private-value') + assert bob.recall('private') == 'private-value' + elif damage == 'missing': + bob._memory.pop('__canary__') + elif damage == 'changed': + bob.remember('__canary__', 'replacement') + else: + bob.canary = alice.canary + bob.remember('__canary__', alice.canary) + + class Driver: + def realize(self, actor, action): + calls.append(action.name) + result = driver.realize(actor, action) + corrupt() + return result + + if not during_run: + corrupt() + result = Runner(world, Driver(), observer, oracle).run(asset) + assert result.verdict is Verdict.INCONCLUSIVE + assert len(calls) == int(during_run) + pack = json.loads(result.evidence.to_json()) + guards = [o['data']['violations'] for o in pack['observations'] + if o['kind'] == 'actor_isolation'] + assert guards[-1] + assert 'private-value' not in result.evidence.to_json() + if damage == 'shared': + assert any('memory store' in v for v in guards[-1]) + packs.append(pack) + assert not classify(packs[0], packs[1]).safe_to_accept + assert not assess_stability(packs).stable diff --git a/workplans/TD-WP-0003-generalise-and-settle.md b/workplans/TD-WP-0003-generalise-and-settle.md index 6c576a1..8c83020 100644 --- a/workplans/TD-WP-0003-generalise-and-settle.md +++ b/workplans/TD-WP-0003-generalise-and-settle.md @@ -353,3 +353,16 @@ evidence despite passing product verdicts, repeated action names with distinct targets, reusable frozen drivers and invalid dispatch. `git diff --check` is clean. Decision: `e8fabf6e-3e5e-424e-9a60-152bf3dec641`. No new task or workplan; T01/T06/T07 remain waiting and this workplan remains blocked. + + +**2026-09-28 guard-integrity follow-up — done.** Oracle now requires actual +boolean predicate results; other values produce INCONCLUSIVE without truth-value +coercion. Actor isolation checks detect shared memory stores, missing/changed +stored markers and invalid/duplicate actor canaries, feeding the existing abort +and acceptance/crystallization gates without disclosing private values. + +Validation: 20 new regressions (16 failed before implementation), 69 focused +checks passed, and the full suite passed **363 tests** in 153.10 seconds. +`git diff --check` is clean. Decision: +`2ecac1c5-5254-4ce9-b092-b47c3e1283a9`. No new task or workplan was opened; +T01/T06/T07 remain waiting and this workplan remains blocked.