test-driver/tests/test_scope_delivery.py

200 lines
8.4 KiB
Python
Raw Normal View History

"""Intent preservation, durable receipts and reusable adversarial variants."""
from concurrent.futures import ThreadPoolExecutor
from dataclasses import replace
import json
import stat
import pytest
from scenarios.alice_bob_carol import build
from scenarios.tenant_lifecycle import build as tenant
from testdriver import Runner, Verdict
from testdriver.classification import classify, Classification
from testdriver.crystallization import assess_stability
from testdriver.evidence import Observation, Stratum
from testdriver.storage import EvidenceStore
from testdriver.variants import substitute
def run(builder=build, store=None):
world, driver, observer, asset, oracle = builder()
return Runner(world, driver, observer, oracle).run(asset, evidence_store=store)
def test_store_roundtrip_retains_lineage_and_admission(tmp_path):
store = EvidenceStore(tmp_path / 'evidence')
results = [run(store=store) for _ in range(3)]
packs = [store.load(result.run_id) for result in results]
assert packs[0] == json.loads(results[0].evidence.to_json())
assert packs[0]['asset']['id'] == build()[3].id
assert packs[0]['run_verdict'] == 'PASS'
assert classify(packs[0], packs[1]).safe_to_accept
assert assess_stability(packs).stable
assert stat.S_IMODE((store.directory / f'{results[0].run_id}.json').stat().st_mode) == 0o600
with pytest.raises(FileExistsError):
store.write(results[0].evidence)
assert not list(store.directory.glob('.evidence-*'))
def test_failed_and_guard_aborted_runs_are_retained(tmp_path):
store = EvidenceStore(tmp_path)
failing = run(lambda: build('M17'), store)
assert store.load(failing.run_id)['run_verdict'] == 'FAIL'
w, d, o, a, oracle = build()
w.cast['bob']._memory = w.cast['alice']._memory
aborted = Runner(w, d, o, oracle).run(a, evidence_store=store)
assert store.load(aborted.run_id)['run_verdict'] == 'INCONCLUSIVE'
@pytest.mark.parametrize('damage', ['checksum', 'schema', 'truncated', 'wrong-id'])
def test_corrupt_receipts_are_rejected(tmp_path, damage):
store = EvidenceStore(tmp_path)
result = run(store=store)
path = tmp_path / f'{result.run_id}.json'
envelope = json.loads(path.read_text())
if damage == 'checksum':
envelope['evidence']['run_verdict'] = 'FAIL'
elif damage == 'schema':
envelope['schema_version'] = 999
elif damage == 'wrong-id':
other = tmp_path / 'other.json'
other.write_text(path.read_text())
with pytest.raises(ValueError):
store.load('other')
return
path.write_text('{' if damage == 'truncated' else json.dumps(envelope))
with pytest.raises(ValueError):
store.load(result.run_id)
@pytest.mark.parametrize('run_id', ['../escape', '/absolute', '.', '', 'a/b'])
def test_store_rejects_unsafe_identifiers_without_writes(tmp_path, run_id):
pack = run().evidence
pack.run_id = run_id
with pytest.raises(ValueError):
EvidenceStore(tmp_path).write(pack)
assert not list(tmp_path.iterdir())
@pytest.mark.parametrize('bad', [object(), float('nan'), {1: 'coerced-key'}])
def test_unserializable_evidence_is_not_silently_stringified(tmp_path, bad):
pack = run().evidence
pack.asset['unsupported'] = bad
with pytest.raises((TypeError, ValueError)):
EvidenceStore(tmp_path).write(pack)
assert not list(tmp_path.iterdir())
def test_concurrent_writers_publish_exactly_one_complete_receipt(tmp_path):
store = EvidenceStore(tmp_path)
pack = run().evidence
def write():
try:
store.write(pack)
return True
except FileExistsError:
return False
with ThreadPoolExecutor(max_workers=4) as workers:
assert sum(workers.map(lambda _: write(), range(4))) == 1
assert store.load(pack.run_id)['run_id'] == pack.run_id
assert not list(tmp_path.glob('.evidence-*'))
def test_runner_propagates_storage_failure(tmp_path):
path = tmp_path / 'file'
path.write_text('not a directory')
with pytest.raises(FileExistsError):
run(store=EvidenceStore(path))
def test_recorded_observations_do_not_change_with_live_nested_objects():
pack = run().evidence
snapshot = {'nested': {'values': [1]}}
pack.record(Observation('custom', Stratum.JUDGMENT, 'observer', None, 'state', snapshot))
snapshot['nested']['values'].append(2)
assert pack.observations[-1].data == {'nested': {'values': [1]}}
def test_variant_preserves_claims_surfaces_and_parent_and_copies_arguments(tmp_path):
world, driver, observer, parent, oracle = build()
step = parent.scenario.steps[1]
variant = substitute(parent, variant_id='carol-read', step_id=step.id,
arguments={'subject_id': 'carol'})
assert variant.parent_id == parent.id
assert variant.scenario.use_case is parent.scenario.use_case
for original, changed in zip(parent.scenario.steps, variant.scenario.steps):
assert original.action.postcondition is changed.action.postcondition
assert original.action.permitted_surfaces == changed.action.permitted_surfaces
assert original.action.args is not changed.action.args
assert step.action.args['subject_id'] == 'bob'
result = Runner(world, driver, observer, oracle).run(variant, evidence_store=EvidenceStore(tmp_path))
assert result.verdict is Verdict.FAIL
assert result.evidence.asset['parent_id'] == parent.id
assert result.judgment('c-carol-denied').verdict is Verdict.FAIL
def test_actor_substitution_applies_to_another_domain_without_new_claims():
world, driver, observer, parent, oracle = tenant()
variant = substitute(parent, variant_id='foreign-creator', step_id='create-a', actor_id='admin-b')
result = Runner(world, driver, observer, oracle).run(variant)
assert result.judgment('tenant-create-a').verdict is Verdict.FAIL
assert variant.scenario.use_case is parent.scenario.use_case
@pytest.mark.parametrize('changes', [
{'step_id': 'absent', 'actor_id': 'bob'},
{'step_id': 's2-grant', 'arguments': {'absent': 1}},
{'step_id': 's2-grant', 'actor_id': ''},
{'step_id': 's2-grant'},
])
def test_invalid_substitutions_do_not_modify_parent(changes):
parent = build()[3]
before = repr(parent)
with pytest.raises(ValueError):
substitute(parent, variant_id='invalid', **changes)
assert repr(parent) == before
@pytest.mark.parametrize('change', ['actor', 'argument', 'surface', 'postcondition', 'order'])
def test_scenario_intent_changes_require_review_even_with_passing_verdicts(change):
baseline = json.loads(run().evidence.to_json())
world, driver, observer, asset, oracle = build()
steps = list(asset.scenario.steps)
step = steps[0]
if change == 'actor':
# Identical authorized operation via another identity, without changing claim definitions.
driver._tokens['carol'] = driver._tokens['alice']
steps[0] = replace(step, actor_id='carol')
elif change == 'argument':
steps[0] = replace(step, action=replace(step.action, args={**step.action.args, 'content': 'changed'}))
elif change == 'surface':
steps[0] = replace(step, action=replace(step.action, permitted_surfaces=frozenset({'api', 'browser'})))
elif change == 'postcondition':
steps[0] = replace(step, action=replace(step.action, postcondition=lambda obs: True))
else:
# A schedule-only assertion-free no-op pair can be reordered without changing outcomes.
steps.extend([replace(step, id='extra-a'), replace(step, id='extra-b')])
steps[-2:] = reversed(steps[-2:])
asset.scenario = replace(asset.scenario, steps=tuple(steps))
result = Runner(world, driver, observer, oracle).run(asset)
pack = json.loads(result.evidence.to_json())
outcome = classify(baseline, pack)
if change != 'order':
assert result.verdict is Verdict.PASS
assert outcome.classification is Classification.INTENT_CHANGED
assert not outcome.safe_to_accept
assert outcome.classification in (Classification.INTENT_CHANGED, Classification.AMBIGUOUS)
def test_legacy_packs_require_fresh_scenario_intent_evidence():
pack = json.loads(run().evidence.to_json())
del pack['scenario_revision']
assert not classify(pack, pack).safe_to_accept
def test_action_order_is_part_of_scenario_revision():
from testdriver.revisions import scenario_revision
scenario = build()[3].scenario
assert scenario_revision(scenario) != scenario_revision(
replace(scenario, steps=tuple(reversed(scenario.steps))))