Reject lossy observations and validate schedules before execution
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
This commit is contained in:
parent
eaf5d348e4
commit
10077edb8c
8 changed files with 184 additions and 21 deletions
88
tests/test_run_preflight_and_json.py
Normal file
88
tests/test_run_preflight_and_json.py
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
"""Reject unsafe run inputs before effects and lossy observation types before judgment."""
|
||||
from dataclasses import replace
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from scenarios.alice_bob_carol import build
|
||||
from testdriver import Runner, Verdict
|
||||
from testdriver.classification import classify
|
||||
from testdriver.storage import EvidenceStore
|
||||
from testdriver.variants import substitute
|
||||
|
||||
|
||||
@pytest.mark.parametrize('bad', [('a', 'b'), {'nested': [('a', 'b')]}, {1: 'value'},
|
||||
float('nan'), float('inf')])
|
||||
def test_lossy_observations_never_produce_a_verdict_that_cannot_be_replayed(tmp_path, bad):
|
||||
w, d, o, a, oracle = build()
|
||||
calls = []
|
||||
case = a.scenario.use_case
|
||||
claim = replace(case.claims[0], predicate=lambda snapshot: snapshot['pair'] == ('a', 'b'))
|
||||
a.scenario = replace(a.scenario, use_case=replace(case, claims=(claim, *case.claims[1:])))
|
||||
|
||||
class Observer:
|
||||
name = o.name
|
||||
def snapshot(self):
|
||||
return {**o.snapshot(), 'pair': bad}
|
||||
|
||||
class Driver:
|
||||
def realize(self, actor, action):
|
||||
calls.append(action.name)
|
||||
return d.realize(actor, action)
|
||||
|
||||
store = EvidenceStore(tmp_path)
|
||||
result = Runner(w, Driver(), Observer(), oracle).run(a, evidence_store=store)
|
||||
assert result.verdict is Verdict.INCONCLUSIVE
|
||||
assert len(calls) == 1
|
||||
assert all(j.verdict is Verdict.INCONCLUSIVE for j in result.judgments)
|
||||
retained = store.load(result.run_id)
|
||||
assert retained['run_verdict'] == 'INCONCLUSIVE'
|
||||
assert any(o['kind'] == 'evidence_failure' for o in retained['observations'])
|
||||
assert not classify(retained, retained).safe_to_accept
|
||||
|
||||
|
||||
def test_json_native_observations_keep_live_and_replayed_verdicts(tmp_path):
|
||||
w, d, o, a, oracle = build()
|
||||
result = Runner(w, d, o, oracle).run(a, evidence_store=EvidenceStore(tmp_path))
|
||||
retained = EvidenceStore(tmp_path).load(result.run_id)
|
||||
snapshots = {o['step_id']: o['data'] for o in retained['observations']
|
||||
if o['kind'] == 'state_snapshot'}
|
||||
assertions = {c.id: c for c in (*a.scenario.use_case.claims, *a.scenario.use_case.invariants)}
|
||||
for verdict in retained['verdicts']:
|
||||
replayed = oracle.judge(assertions[verdict['assertion_id']], snapshots[verdict['step_id']], verdict['step_id'])
|
||||
assert replayed.verdict.value == verdict['verdict']
|
||||
assert result.verdict is Verdict.PASS
|
||||
|
||||
|
||||
@pytest.mark.parametrize('bad', [('a', 'b'), {1: 'value'}])
|
||||
def test_serializer_does_not_silently_convert_unsupported_types(bad):
|
||||
w, d, o, a, oracle = build()
|
||||
pack = Runner(w, d, o, oracle).run(a).evidence
|
||||
pack.asset['invalid'] = bad
|
||||
with pytest.raises(TypeError):
|
||||
pack.to_json()
|
||||
|
||||
|
||||
@pytest.mark.parametrize('damage', ['unknown-actor', 'duplicate-step', 'empty-step', 'cast-mismatch'])
|
||||
def test_entire_schedule_is_validated_before_any_side_effect(tmp_path, damage):
|
||||
w, d, o, a, oracle = build()
|
||||
calls = []
|
||||
if damage == 'unknown-actor':
|
||||
a = substitute(a, variant_id='typo', step_id=a.scenario.steps[-1].id, actor_id='unknown')
|
||||
elif damage == 'cast-mismatch':
|
||||
w.cast['alice'].id = 'different'
|
||||
else:
|
||||
steps = list(a.scenario.steps)
|
||||
steps[-1] = replace(steps[-1], id='' if damage == 'empty-step' else steps[0].id)
|
||||
a.scenario = replace(a.scenario, steps=tuple(steps))
|
||||
|
||||
class Driver:
|
||||
def realize(self, actor, action):
|
||||
calls.append(action.name)
|
||||
return d.realize(actor, action)
|
||||
|
||||
with pytest.raises(ValueError, match='scenario|actor|step|cast'):
|
||||
Runner(w, Driver(), o, oracle).run(a, evidence_store=EvidenceStore(tmp_path))
|
||||
assert calls == []
|
||||
assert not w.sut.resources
|
||||
assert not list(tmp_path.iterdir())
|
||||
Loading…
Add table
Add a link
Reference in a new issue