test-driver/tests/test_scope_delivery.py
tegwick eaf5d348e4 Align scope with intent and add durable evidence and bounded variants
Assistant: codex
Assistant-Model: gpt-6-astra
Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
2026-09-28 14:31:22 +02:00

199 lines
8.4 KiB
Python

"""Intent preservation, durable receipts and reusable adversarial variants."""
from concurrent.futures import ThreadPoolExecutor
from dataclasses import replace
import json
import stat
import pytest
from scenarios.alice_bob_carol import build
from scenarios.tenant_lifecycle import build as tenant
from testdriver import Runner, Verdict
from testdriver.classification import classify, Classification
from testdriver.crystallization import assess_stability
from testdriver.evidence import Observation, Stratum
from testdriver.storage import EvidenceStore
from testdriver.variants import substitute
def run(builder=build, store=None):
world, driver, observer, asset, oracle = builder()
return Runner(world, driver, observer, oracle).run(asset, evidence_store=store)
def test_store_roundtrip_retains_lineage_and_admission(tmp_path):
store = EvidenceStore(tmp_path / 'evidence')
results = [run(store=store) for _ in range(3)]
packs = [store.load(result.run_id) for result in results]
assert packs[0] == json.loads(results[0].evidence.to_json())
assert packs[0]['asset']['id'] == build()[3].id
assert packs[0]['run_verdict'] == 'PASS'
assert classify(packs[0], packs[1]).safe_to_accept
assert assess_stability(packs).stable
assert stat.S_IMODE((store.directory / f'{results[0].run_id}.json').stat().st_mode) == 0o600
with pytest.raises(FileExistsError):
store.write(results[0].evidence)
assert not list(store.directory.glob('.evidence-*'))
def test_failed_and_guard_aborted_runs_are_retained(tmp_path):
store = EvidenceStore(tmp_path)
failing = run(lambda: build('M17'), store)
assert store.load(failing.run_id)['run_verdict'] == 'FAIL'
w, d, o, a, oracle = build()
w.cast['bob']._memory = w.cast['alice']._memory
aborted = Runner(w, d, o, oracle).run(a, evidence_store=store)
assert store.load(aborted.run_id)['run_verdict'] == 'INCONCLUSIVE'
@pytest.mark.parametrize('damage', ['checksum', 'schema', 'truncated', 'wrong-id'])
def test_corrupt_receipts_are_rejected(tmp_path, damage):
store = EvidenceStore(tmp_path)
result = run(store=store)
path = tmp_path / f'{result.run_id}.json'
envelope = json.loads(path.read_text())
if damage == 'checksum':
envelope['evidence']['run_verdict'] = 'FAIL'
elif damage == 'schema':
envelope['schema_version'] = 999
elif damage == 'wrong-id':
other = tmp_path / 'other.json'
other.write_text(path.read_text())
with pytest.raises(ValueError):
store.load('other')
return
path.write_text('{' if damage == 'truncated' else json.dumps(envelope))
with pytest.raises(ValueError):
store.load(result.run_id)
@pytest.mark.parametrize('run_id', ['../escape', '/absolute', '.', '', 'a/b'])
def test_store_rejects_unsafe_identifiers_without_writes(tmp_path, run_id):
pack = run().evidence
pack.run_id = run_id
with pytest.raises(ValueError):
EvidenceStore(tmp_path).write(pack)
assert not list(tmp_path.iterdir())
@pytest.mark.parametrize('bad', [object(), float('nan'), {1: 'coerced-key'}])
def test_unserializable_evidence_is_not_silently_stringified(tmp_path, bad):
pack = run().evidence
pack.asset['unsupported'] = bad
with pytest.raises((TypeError, ValueError)):
EvidenceStore(tmp_path).write(pack)
assert not list(tmp_path.iterdir())
def test_concurrent_writers_publish_exactly_one_complete_receipt(tmp_path):
store = EvidenceStore(tmp_path)
pack = run().evidence
def write():
try:
store.write(pack)
return True
except FileExistsError:
return False
with ThreadPoolExecutor(max_workers=4) as workers:
assert sum(workers.map(lambda _: write(), range(4))) == 1
assert store.load(pack.run_id)['run_id'] == pack.run_id
assert not list(tmp_path.glob('.evidence-*'))
def test_runner_propagates_storage_failure(tmp_path):
path = tmp_path / 'file'
path.write_text('not a directory')
with pytest.raises(FileExistsError):
run(store=EvidenceStore(path))
def test_recorded_observations_do_not_change_with_live_nested_objects():
pack = run().evidence
snapshot = {'nested': {'values': [1]}}
pack.record(Observation('custom', Stratum.JUDGMENT, 'observer', None, 'state', snapshot))
snapshot['nested']['values'].append(2)
assert pack.observations[-1].data == {'nested': {'values': [1]}}
def test_variant_preserves_claims_surfaces_and_parent_and_copies_arguments(tmp_path):
world, driver, observer, parent, oracle = build()
step = parent.scenario.steps[1]
variant = substitute(parent, variant_id='carol-read', step_id=step.id,
arguments={'subject_id': 'carol'})
assert variant.parent_id == parent.id
assert variant.scenario.use_case is parent.scenario.use_case
for original, changed in zip(parent.scenario.steps, variant.scenario.steps):
assert original.action.postcondition is changed.action.postcondition
assert original.action.permitted_surfaces == changed.action.permitted_surfaces
assert original.action.args is not changed.action.args
assert step.action.args['subject_id'] == 'bob'
result = Runner(world, driver, observer, oracle).run(variant, evidence_store=EvidenceStore(tmp_path))
assert result.verdict is Verdict.FAIL
assert result.evidence.asset['parent_id'] == parent.id
assert result.judgment('c-carol-denied').verdict is Verdict.FAIL
def test_actor_substitution_applies_to_another_domain_without_new_claims():
world, driver, observer, parent, oracle = tenant()
variant = substitute(parent, variant_id='foreign-creator', step_id='create-a', actor_id='admin-b')
result = Runner(world, driver, observer, oracle).run(variant)
assert result.judgment('tenant-create-a').verdict is Verdict.FAIL
assert variant.scenario.use_case is parent.scenario.use_case
@pytest.mark.parametrize('changes', [
{'step_id': 'absent', 'actor_id': 'bob'},
{'step_id': 's2-grant', 'arguments': {'absent': 1}},
{'step_id': 's2-grant', 'actor_id': ''},
{'step_id': 's2-grant'},
])
def test_invalid_substitutions_do_not_modify_parent(changes):
parent = build()[3]
before = repr(parent)
with pytest.raises(ValueError):
substitute(parent, variant_id='invalid', **changes)
assert repr(parent) == before
@pytest.mark.parametrize('change', ['actor', 'argument', 'surface', 'postcondition', 'order'])
def test_scenario_intent_changes_require_review_even_with_passing_verdicts(change):
baseline = json.loads(run().evidence.to_json())
world, driver, observer, asset, oracle = build()
steps = list(asset.scenario.steps)
step = steps[0]
if change == 'actor':
# Identical authorized operation via another identity, without changing claim definitions.
driver._tokens['carol'] = driver._tokens['alice']
steps[0] = replace(step, actor_id='carol')
elif change == 'argument':
steps[0] = replace(step, action=replace(step.action, args={**step.action.args, 'content': 'changed'}))
elif change == 'surface':
steps[0] = replace(step, action=replace(step.action, permitted_surfaces=frozenset({'api', 'browser'})))
elif change == 'postcondition':
steps[0] = replace(step, action=replace(step.action, postcondition=lambda obs: True))
else:
# A schedule-only assertion-free no-op pair can be reordered without changing outcomes.
steps.extend([replace(step, id='extra-a'), replace(step, id='extra-b')])
steps[-2:] = reversed(steps[-2:])
asset.scenario = replace(asset.scenario, steps=tuple(steps))
result = Runner(world, driver, observer, oracle).run(asset)
pack = json.loads(result.evidence.to_json())
outcome = classify(baseline, pack)
if change != 'order':
assert result.verdict is Verdict.PASS
assert outcome.classification is Classification.INTENT_CHANGED
assert not outcome.safe_to_accept
assert outcome.classification in (Classification.INTENT_CHANGED, Classification.AMBIGUOUS)
def test_legacy_packs_require_fresh_scenario_intent_evidence():
pack = json.loads(run().evidence.to_json())
del pack['scenario_revision']
assert not classify(pack, pack).safe_to_accept
def test_action_order_is_part_of_scenario_revision():
from testdriver.revisions import scenario_revision
scenario = build()[3].scenario
assert scenario_revision(scenario) != scenario_revision(
replace(scenario, steps=tuple(reversed(scenario.steps))))