Align scope with intent and add durable evidence and bounded variants
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
parent
8822480b6e
commit
eaf5d348e4
14 changed files with 767 additions and 92 deletions
199
tests/test_scope_delivery.py
Normal file
199
tests/test_scope_delivery.py
Normal file
|
|
@ -0,0 +1,199 @@
|
|||
"""Intent preservation, durable receipts and reusable adversarial variants."""
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from dataclasses import replace
|
||||
import json
|
||||
import stat
|
||||
|
||||
import pytest
|
||||
|
||||
from scenarios.alice_bob_carol import build
|
||||
from scenarios.tenant_lifecycle import build as tenant
|
||||
from testdriver import Runner, Verdict
|
||||
from testdriver.classification import classify, Classification
|
||||
from testdriver.crystallization import assess_stability
|
||||
from testdriver.evidence import Observation, Stratum
|
||||
from testdriver.storage import EvidenceStore
|
||||
from testdriver.variants import substitute
|
||||
|
||||
|
||||
def run(builder=build, store=None):
|
||||
world, driver, observer, asset, oracle = builder()
|
||||
return Runner(world, driver, observer, oracle).run(asset, evidence_store=store)
|
||||
|
||||
|
||||
def test_store_roundtrip_retains_lineage_and_admission(tmp_path):
|
||||
store = EvidenceStore(tmp_path / 'evidence')
|
||||
results = [run(store=store) for _ in range(3)]
|
||||
packs = [store.load(result.run_id) for result in results]
|
||||
assert packs[0] == json.loads(results[0].evidence.to_json())
|
||||
assert packs[0]['asset']['id'] == build()[3].id
|
||||
assert packs[0]['run_verdict'] == 'PASS'
|
||||
assert classify(packs[0], packs[1]).safe_to_accept
|
||||
assert assess_stability(packs).stable
|
||||
assert stat.S_IMODE((store.directory / f'{results[0].run_id}.json').stat().st_mode) == 0o600
|
||||
with pytest.raises(FileExistsError):
|
||||
store.write(results[0].evidence)
|
||||
assert not list(store.directory.glob('.evidence-*'))
|
||||
|
||||
|
||||
def test_failed_and_guard_aborted_runs_are_retained(tmp_path):
|
||||
store = EvidenceStore(tmp_path)
|
||||
failing = run(lambda: build('M17'), store)
|
||||
assert store.load(failing.run_id)['run_verdict'] == 'FAIL'
|
||||
w, d, o, a, oracle = build()
|
||||
w.cast['bob']._memory = w.cast['alice']._memory
|
||||
aborted = Runner(w, d, o, oracle).run(a, evidence_store=store)
|
||||
assert store.load(aborted.run_id)['run_verdict'] == 'INCONCLUSIVE'
|
||||
|
||||
|
||||
@pytest.mark.parametrize('damage', ['checksum', 'schema', 'truncated', 'wrong-id'])
|
||||
def test_corrupt_receipts_are_rejected(tmp_path, damage):
|
||||
store = EvidenceStore(tmp_path)
|
||||
result = run(store=store)
|
||||
path = tmp_path / f'{result.run_id}.json'
|
||||
envelope = json.loads(path.read_text())
|
||||
if damage == 'checksum':
|
||||
envelope['evidence']['run_verdict'] = 'FAIL'
|
||||
elif damage == 'schema':
|
||||
envelope['schema_version'] = 999
|
||||
elif damage == 'wrong-id':
|
||||
other = tmp_path / 'other.json'
|
||||
other.write_text(path.read_text())
|
||||
with pytest.raises(ValueError):
|
||||
store.load('other')
|
||||
return
|
||||
path.write_text('{' if damage == 'truncated' else json.dumps(envelope))
|
||||
with pytest.raises(ValueError):
|
||||
store.load(result.run_id)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('run_id', ['../escape', '/absolute', '.', '', 'a/b'])
|
||||
def test_store_rejects_unsafe_identifiers_without_writes(tmp_path, run_id):
|
||||
pack = run().evidence
|
||||
pack.run_id = run_id
|
||||
with pytest.raises(ValueError):
|
||||
EvidenceStore(tmp_path).write(pack)
|
||||
assert not list(tmp_path.iterdir())
|
||||
|
||||
|
||||
@pytest.mark.parametrize('bad', [object(), float('nan'), {1: 'coerced-key'}])
|
||||
def test_unserializable_evidence_is_not_silently_stringified(tmp_path, bad):
|
||||
pack = run().evidence
|
||||
pack.asset['unsupported'] = bad
|
||||
with pytest.raises((TypeError, ValueError)):
|
||||
EvidenceStore(tmp_path).write(pack)
|
||||
assert not list(tmp_path.iterdir())
|
||||
|
||||
|
||||
def test_concurrent_writers_publish_exactly_one_complete_receipt(tmp_path):
|
||||
store = EvidenceStore(tmp_path)
|
||||
pack = run().evidence
|
||||
def write():
|
||||
try:
|
||||
store.write(pack)
|
||||
return True
|
||||
except FileExistsError:
|
||||
return False
|
||||
with ThreadPoolExecutor(max_workers=4) as workers:
|
||||
assert sum(workers.map(lambda _: write(), range(4))) == 1
|
||||
assert store.load(pack.run_id)['run_id'] == pack.run_id
|
||||
assert not list(tmp_path.glob('.evidence-*'))
|
||||
|
||||
|
||||
def test_runner_propagates_storage_failure(tmp_path):
|
||||
path = tmp_path / 'file'
|
||||
path.write_text('not a directory')
|
||||
with pytest.raises(FileExistsError):
|
||||
run(store=EvidenceStore(path))
|
||||
|
||||
|
||||
def test_recorded_observations_do_not_change_with_live_nested_objects():
|
||||
pack = run().evidence
|
||||
snapshot = {'nested': {'values': [1]}}
|
||||
pack.record(Observation('custom', Stratum.JUDGMENT, 'observer', None, 'state', snapshot))
|
||||
snapshot['nested']['values'].append(2)
|
||||
assert pack.observations[-1].data == {'nested': {'values': [1]}}
|
||||
|
||||
|
||||
def test_variant_preserves_claims_surfaces_and_parent_and_copies_arguments(tmp_path):
|
||||
world, driver, observer, parent, oracle = build()
|
||||
step = parent.scenario.steps[1]
|
||||
variant = substitute(parent, variant_id='carol-read', step_id=step.id,
|
||||
arguments={'subject_id': 'carol'})
|
||||
assert variant.parent_id == parent.id
|
||||
assert variant.scenario.use_case is parent.scenario.use_case
|
||||
for original, changed in zip(parent.scenario.steps, variant.scenario.steps):
|
||||
assert original.action.postcondition is changed.action.postcondition
|
||||
assert original.action.permitted_surfaces == changed.action.permitted_surfaces
|
||||
assert original.action.args is not changed.action.args
|
||||
assert step.action.args['subject_id'] == 'bob'
|
||||
result = Runner(world, driver, observer, oracle).run(variant, evidence_store=EvidenceStore(tmp_path))
|
||||
assert result.verdict is Verdict.FAIL
|
||||
assert result.evidence.asset['parent_id'] == parent.id
|
||||
assert result.judgment('c-carol-denied').verdict is Verdict.FAIL
|
||||
|
||||
|
||||
def test_actor_substitution_applies_to_another_domain_without_new_claims():
|
||||
world, driver, observer, parent, oracle = tenant()
|
||||
variant = substitute(parent, variant_id='foreign-creator', step_id='create-a', actor_id='admin-b')
|
||||
result = Runner(world, driver, observer, oracle).run(variant)
|
||||
assert result.judgment('tenant-create-a').verdict is Verdict.FAIL
|
||||
assert variant.scenario.use_case is parent.scenario.use_case
|
||||
|
||||
|
||||
@pytest.mark.parametrize('changes', [
|
||||
{'step_id': 'absent', 'actor_id': 'bob'},
|
||||
{'step_id': 's2-grant', 'arguments': {'absent': 1}},
|
||||
{'step_id': 's2-grant', 'actor_id': ''},
|
||||
{'step_id': 's2-grant'},
|
||||
])
|
||||
def test_invalid_substitutions_do_not_modify_parent(changes):
|
||||
parent = build()[3]
|
||||
before = repr(parent)
|
||||
with pytest.raises(ValueError):
|
||||
substitute(parent, variant_id='invalid', **changes)
|
||||
assert repr(parent) == before
|
||||
|
||||
|
||||
@pytest.mark.parametrize('change', ['actor', 'argument', 'surface', 'postcondition', 'order'])
|
||||
def test_scenario_intent_changes_require_review_even_with_passing_verdicts(change):
|
||||
baseline = json.loads(run().evidence.to_json())
|
||||
world, driver, observer, asset, oracle = build()
|
||||
steps = list(asset.scenario.steps)
|
||||
step = steps[0]
|
||||
if change == 'actor':
|
||||
# Identical authorized operation via another identity, without changing claim definitions.
|
||||
driver._tokens['carol'] = driver._tokens['alice']
|
||||
steps[0] = replace(step, actor_id='carol')
|
||||
elif change == 'argument':
|
||||
steps[0] = replace(step, action=replace(step.action, args={**step.action.args, 'content': 'changed'}))
|
||||
elif change == 'surface':
|
||||
steps[0] = replace(step, action=replace(step.action, permitted_surfaces=frozenset({'api', 'browser'})))
|
||||
elif change == 'postcondition':
|
||||
steps[0] = replace(step, action=replace(step.action, postcondition=lambda obs: True))
|
||||
else:
|
||||
# A schedule-only assertion-free no-op pair can be reordered without changing outcomes.
|
||||
steps.extend([replace(step, id='extra-a'), replace(step, id='extra-b')])
|
||||
steps[-2:] = reversed(steps[-2:])
|
||||
asset.scenario = replace(asset.scenario, steps=tuple(steps))
|
||||
result = Runner(world, driver, observer, oracle).run(asset)
|
||||
pack = json.loads(result.evidence.to_json())
|
||||
outcome = classify(baseline, pack)
|
||||
if change != 'order':
|
||||
assert result.verdict is Verdict.PASS
|
||||
assert outcome.classification is Classification.INTENT_CHANGED
|
||||
assert not outcome.safe_to_accept
|
||||
assert outcome.classification in (Classification.INTENT_CHANGED, Classification.AMBIGUOUS)
|
||||
|
||||
|
||||
def test_legacy_packs_require_fresh_scenario_intent_evidence():
|
||||
pack = json.loads(run().evidence.to_json())
|
||||
del pack['scenario_revision']
|
||||
assert not classify(pack, pack).safe_to_accept
|
||||
|
||||
|
||||
def test_action_order_is_part_of_scenario_revision():
|
||||
from testdriver.revisions import scenario_revision
|
||||
scenario = build()[3].scenario
|
||||
assert scenario_revision(scenario) != scenario_revision(
|
||||
replace(scenario, steps=tuple(reversed(scenario.steps))))
|
||||
Loading…
Add table
Add a link
Reference in a new issue