Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
199 lines
8.4 KiB
Python
199 lines
8.4 KiB
Python
"""Intent preservation, durable receipts and reusable adversarial variants."""
|
|
from concurrent.futures import ThreadPoolExecutor
|
|
from dataclasses import replace
|
|
import json
|
|
import stat
|
|
|
|
import pytest
|
|
|
|
from scenarios.alice_bob_carol import build
|
|
from scenarios.tenant_lifecycle import build as tenant
|
|
from testdriver import Runner, Verdict
|
|
from testdriver.classification import classify, Classification
|
|
from testdriver.crystallization import assess_stability
|
|
from testdriver.evidence import Observation, Stratum
|
|
from testdriver.storage import EvidenceStore
|
|
from testdriver.variants import substitute
|
|
|
|
|
|
def run(builder=build, store=None):
|
|
world, driver, observer, asset, oracle = builder()
|
|
return Runner(world, driver, observer, oracle).run(asset, evidence_store=store)
|
|
|
|
|
|
def test_store_roundtrip_retains_lineage_and_admission(tmp_path):
|
|
store = EvidenceStore(tmp_path / 'evidence')
|
|
results = [run(store=store) for _ in range(3)]
|
|
packs = [store.load(result.run_id) for result in results]
|
|
assert packs[0] == json.loads(results[0].evidence.to_json())
|
|
assert packs[0]['asset']['id'] == build()[3].id
|
|
assert packs[0]['run_verdict'] == 'PASS'
|
|
assert classify(packs[0], packs[1]).safe_to_accept
|
|
assert assess_stability(packs).stable
|
|
assert stat.S_IMODE((store.directory / f'{results[0].run_id}.json').stat().st_mode) == 0o600
|
|
with pytest.raises(FileExistsError):
|
|
store.write(results[0].evidence)
|
|
assert not list(store.directory.glob('.evidence-*'))
|
|
|
|
|
|
def test_failed_and_guard_aborted_runs_are_retained(tmp_path):
|
|
store = EvidenceStore(tmp_path)
|
|
failing = run(lambda: build('M17'), store)
|
|
assert store.load(failing.run_id)['run_verdict'] == 'FAIL'
|
|
w, d, o, a, oracle = build()
|
|
w.cast['bob']._memory = w.cast['alice']._memory
|
|
aborted = Runner(w, d, o, oracle).run(a, evidence_store=store)
|
|
assert store.load(aborted.run_id)['run_verdict'] == 'INCONCLUSIVE'
|
|
|
|
|
|
@pytest.mark.parametrize('damage', ['checksum', 'schema', 'truncated', 'wrong-id'])
|
|
def test_corrupt_receipts_are_rejected(tmp_path, damage):
|
|
store = EvidenceStore(tmp_path)
|
|
result = run(store=store)
|
|
path = tmp_path / f'{result.run_id}.json'
|
|
envelope = json.loads(path.read_text())
|
|
if damage == 'checksum':
|
|
envelope['evidence']['run_verdict'] = 'FAIL'
|
|
elif damage == 'schema':
|
|
envelope['schema_version'] = 999
|
|
elif damage == 'wrong-id':
|
|
other = tmp_path / 'other.json'
|
|
other.write_text(path.read_text())
|
|
with pytest.raises(ValueError):
|
|
store.load('other')
|
|
return
|
|
path.write_text('{' if damage == 'truncated' else json.dumps(envelope))
|
|
with pytest.raises(ValueError):
|
|
store.load(result.run_id)
|
|
|
|
|
|
@pytest.mark.parametrize('run_id', ['../escape', '/absolute', '.', '', 'a/b'])
|
|
def test_store_rejects_unsafe_identifiers_without_writes(tmp_path, run_id):
|
|
pack = run().evidence
|
|
pack.run_id = run_id
|
|
with pytest.raises(ValueError):
|
|
EvidenceStore(tmp_path).write(pack)
|
|
assert not list(tmp_path.iterdir())
|
|
|
|
|
|
@pytest.mark.parametrize('bad', [object(), float('nan'), {1: 'coerced-key'}])
|
|
def test_unserializable_evidence_is_not_silently_stringified(tmp_path, bad):
|
|
pack = run().evidence
|
|
pack.asset['unsupported'] = bad
|
|
with pytest.raises((TypeError, ValueError)):
|
|
EvidenceStore(tmp_path).write(pack)
|
|
assert not list(tmp_path.iterdir())
|
|
|
|
|
|
def test_concurrent_writers_publish_exactly_one_complete_receipt(tmp_path):
|
|
store = EvidenceStore(tmp_path)
|
|
pack = run().evidence
|
|
def write():
|
|
try:
|
|
store.write(pack)
|
|
return True
|
|
except FileExistsError:
|
|
return False
|
|
with ThreadPoolExecutor(max_workers=4) as workers:
|
|
assert sum(workers.map(lambda _: write(), range(4))) == 1
|
|
assert store.load(pack.run_id)['run_id'] == pack.run_id
|
|
assert not list(tmp_path.glob('.evidence-*'))
|
|
|
|
|
|
def test_runner_propagates_storage_failure(tmp_path):
|
|
path = tmp_path / 'file'
|
|
path.write_text('not a directory')
|
|
with pytest.raises(FileExistsError):
|
|
run(store=EvidenceStore(path))
|
|
|
|
|
|
def test_recorded_observations_do_not_change_with_live_nested_objects():
|
|
pack = run().evidence
|
|
snapshot = {'nested': {'values': [1]}}
|
|
pack.record(Observation('custom', Stratum.JUDGMENT, 'observer', None, 'state', snapshot))
|
|
snapshot['nested']['values'].append(2)
|
|
assert pack.observations[-1].data == {'nested': {'values': [1]}}
|
|
|
|
|
|
def test_variant_preserves_claims_surfaces_and_parent_and_copies_arguments(tmp_path):
|
|
world, driver, observer, parent, oracle = build()
|
|
step = parent.scenario.steps[1]
|
|
variant = substitute(parent, variant_id='carol-read', step_id=step.id,
|
|
arguments={'subject_id': 'carol'})
|
|
assert variant.parent_id == parent.id
|
|
assert variant.scenario.use_case is parent.scenario.use_case
|
|
for original, changed in zip(parent.scenario.steps, variant.scenario.steps):
|
|
assert original.action.postcondition is changed.action.postcondition
|
|
assert original.action.permitted_surfaces == changed.action.permitted_surfaces
|
|
assert original.action.args is not changed.action.args
|
|
assert step.action.args['subject_id'] == 'bob'
|
|
result = Runner(world, driver, observer, oracle).run(variant, evidence_store=EvidenceStore(tmp_path))
|
|
assert result.verdict is Verdict.FAIL
|
|
assert result.evidence.asset['parent_id'] == parent.id
|
|
assert result.judgment('c-carol-denied').verdict is Verdict.FAIL
|
|
|
|
|
|
def test_actor_substitution_applies_to_another_domain_without_new_claims():
|
|
world, driver, observer, parent, oracle = tenant()
|
|
variant = substitute(parent, variant_id='foreign-creator', step_id='create-a', actor_id='admin-b')
|
|
result = Runner(world, driver, observer, oracle).run(variant)
|
|
assert result.judgment('tenant-create-a').verdict is Verdict.FAIL
|
|
assert variant.scenario.use_case is parent.scenario.use_case
|
|
|
|
|
|
@pytest.mark.parametrize('changes', [
|
|
{'step_id': 'absent', 'actor_id': 'bob'},
|
|
{'step_id': 's2-grant', 'arguments': {'absent': 1}},
|
|
{'step_id': 's2-grant', 'actor_id': ''},
|
|
{'step_id': 's2-grant'},
|
|
])
|
|
def test_invalid_substitutions_do_not_modify_parent(changes):
|
|
parent = build()[3]
|
|
before = repr(parent)
|
|
with pytest.raises(ValueError):
|
|
substitute(parent, variant_id='invalid', **changes)
|
|
assert repr(parent) == before
|
|
|
|
|
|
@pytest.mark.parametrize('change', ['actor', 'argument', 'surface', 'postcondition', 'order'])
|
|
def test_scenario_intent_changes_require_review_even_with_passing_verdicts(change):
|
|
baseline = json.loads(run().evidence.to_json())
|
|
world, driver, observer, asset, oracle = build()
|
|
steps = list(asset.scenario.steps)
|
|
step = steps[0]
|
|
if change == 'actor':
|
|
# Identical authorized operation via another identity, without changing claim definitions.
|
|
driver._tokens['carol'] = driver._tokens['alice']
|
|
steps[0] = replace(step, actor_id='carol')
|
|
elif change == 'argument':
|
|
steps[0] = replace(step, action=replace(step.action, args={**step.action.args, 'content': 'changed'}))
|
|
elif change == 'surface':
|
|
steps[0] = replace(step, action=replace(step.action, permitted_surfaces=frozenset({'api', 'browser'})))
|
|
elif change == 'postcondition':
|
|
steps[0] = replace(step, action=replace(step.action, postcondition=lambda obs: True))
|
|
else:
|
|
# A schedule-only assertion-free no-op pair can be reordered without changing outcomes.
|
|
steps.extend([replace(step, id='extra-a'), replace(step, id='extra-b')])
|
|
steps[-2:] = reversed(steps[-2:])
|
|
asset.scenario = replace(asset.scenario, steps=tuple(steps))
|
|
result = Runner(world, driver, observer, oracle).run(asset)
|
|
pack = json.loads(result.evidence.to_json())
|
|
outcome = classify(baseline, pack)
|
|
if change != 'order':
|
|
assert result.verdict is Verdict.PASS
|
|
assert outcome.classification is Classification.INTENT_CHANGED
|
|
assert not outcome.safe_to_accept
|
|
assert outcome.classification in (Classification.INTENT_CHANGED, Classification.AMBIGUOUS)
|
|
|
|
|
|
def test_legacy_packs_require_fresh_scenario_intent_evidence():
|
|
pack = json.loads(run().evidence.to_json())
|
|
del pack['scenario_revision']
|
|
assert not classify(pack, pack).safe_to_accept
|
|
|
|
|
|
def test_action_order_is_part_of_scenario_revision():
|
|
from testdriver.revisions import scenario_revision
|
|
scenario = build()[3].scenario
|
|
assert scenario_revision(scenario) != scenario_revision(
|
|
replace(scenario, steps=tuple(reversed(scenario.steps))))
|