test-driver/tests/test_run_preflight_and_json.py
tegwick 10077edb8c Reject lossy observations and validate schedules before execution
Assistant: codex
Assistant-Model: gpt-6-astra
Assistant-Session: 01a0e76f-be98-7ae3-965d-e0b31290a4c4
2026-09-28 14:50:27 +02:00

88 lines
3.6 KiB
Python

"""Reject unsafe run inputs before effects and lossy observation types before judgment."""
from dataclasses import replace
import json
import pytest
from scenarios.alice_bob_carol import build
from testdriver import Runner, Verdict
from testdriver.classification import classify
from testdriver.storage import EvidenceStore
from testdriver.variants import substitute
@pytest.mark.parametrize('bad', [('a', 'b'), {'nested': [('a', 'b')]}, {1: 'value'},
float('nan'), float('inf')])
def test_lossy_observations_never_produce_a_verdict_that_cannot_be_replayed(tmp_path, bad):
w, d, o, a, oracle = build()
calls = []
case = a.scenario.use_case
claim = replace(case.claims[0], predicate=lambda snapshot: snapshot['pair'] == ('a', 'b'))
a.scenario = replace(a.scenario, use_case=replace(case, claims=(claim, *case.claims[1:])))
class Observer:
name = o.name
def snapshot(self):
return {**o.snapshot(), 'pair': bad}
class Driver:
def realize(self, actor, action):
calls.append(action.name)
return d.realize(actor, action)
store = EvidenceStore(tmp_path)
result = Runner(w, Driver(), Observer(), oracle).run(a, evidence_store=store)
assert result.verdict is Verdict.INCONCLUSIVE
assert len(calls) == 1
assert all(j.verdict is Verdict.INCONCLUSIVE for j in result.judgments)
retained = store.load(result.run_id)
assert retained['run_verdict'] == 'INCONCLUSIVE'
assert any(o['kind'] == 'evidence_failure' for o in retained['observations'])
assert not classify(retained, retained).safe_to_accept
def test_json_native_observations_keep_live_and_replayed_verdicts(tmp_path):
w, d, o, a, oracle = build()
result = Runner(w, d, o, oracle).run(a, evidence_store=EvidenceStore(tmp_path))
retained = EvidenceStore(tmp_path).load(result.run_id)
snapshots = {o['step_id']: o['data'] for o in retained['observations']
if o['kind'] == 'state_snapshot'}
assertions = {c.id: c for c in (*a.scenario.use_case.claims, *a.scenario.use_case.invariants)}
for verdict in retained['verdicts']:
replayed = oracle.judge(assertions[verdict['assertion_id']], snapshots[verdict['step_id']], verdict['step_id'])
assert replayed.verdict.value == verdict['verdict']
assert result.verdict is Verdict.PASS
@pytest.mark.parametrize('bad', [('a', 'b'), {1: 'value'}])
def test_serializer_does_not_silently_convert_unsupported_types(bad):
w, d, o, a, oracle = build()
pack = Runner(w, d, o, oracle).run(a).evidence
pack.asset['invalid'] = bad
with pytest.raises(TypeError):
pack.to_json()
@pytest.mark.parametrize('damage', ['unknown-actor', 'duplicate-step', 'empty-step', 'cast-mismatch'])
def test_entire_schedule_is_validated_before_any_side_effect(tmp_path, damage):
w, d, o, a, oracle = build()
calls = []
if damage == 'unknown-actor':
a = substitute(a, variant_id='typo', step_id=a.scenario.steps[-1].id, actor_id='unknown')
elif damage == 'cast-mismatch':
w.cast['alice'].id = 'different'
else:
steps = list(a.scenario.steps)
steps[-1] = replace(steps[-1], id='' if damage == 'empty-step' else steps[0].id)
a.scenario = replace(a.scenario, steps=tuple(steps))
class Driver:
def realize(self, actor, action):
calls.append(action.name)
return d.realize(actor, action)
with pytest.raises(ValueError, match='scenario|actor|step|cast'):
Runner(w, Driver(), o, oracle).run(a, evidence_store=EvidenceStore(tmp_path))
assert calls == []
assert not w.sut.resources
assert not list(tmp_path.iterdir())