T07: agentic realization over a stdlib browser surface
Two decisions taken with the operator: stdlib HTML driver instead of Playwright (F-0004), and a deterministic discovery runtime instead of a live model. Both sit behind interfaces so the alternatives drop in later. - html.py: stdlib DOM parse and query - agentic.py: DiscoveryRuntime (agentic arm, ignores data-td by construction) and RecordedSelectorRuntime (control arm, uses the strongest identifier the page offers) - browser.py: per-actor sessions over real HTTP, constructed per call so no actor inherits another's connection state - cost/nondeterminism metrics recorded from the first run F-0005 (CONCEPT_DRIFT): the H-001 result is a narrowing. Where test ids are preserved, discovery 9/9 and recorded selectors 9/9 - the semantic action buys nothing. Where they are dropped, discovery 2/3 and recorded 0/3. The concept model presents semantic actions as generally superior; the evidence says conditionally superior. M21 and M22 added mid-task: the deciding side of the axis was N=1. M22 (field names renamed) defeats the heuristic and is the first concrete evidence that a live model would add capability, not just cost. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Assistant: claude-code Assistant-Model: opus Assistant-Process: 1629012@bnt-lap001 Assistant-Session: 78d4fb13-8a1e-474b-87a3-9b9261c49a39
This commit is contained in:
parent
925ff2dd91
commit
44faf3de8e
23 changed files with 1008 additions and 9 deletions
175
tests/test_agentic_realization.py
Normal file
175
tests/test_agentic_realization.py
Normal file
|
|
@ -0,0 +1,175 @@
|
|||
"""Agentic realization of one semantic action, against a real HTTP surface.
|
||||
|
||||
The agentic part is confined to *finding the control*. Setup, revocation and
|
||||
every oracle remain deterministic, so a failure here is a failure to realize —
|
||||
never a failure to judge.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from testdriver import Runner, SemanticAction, Stratum, Verdict
|
||||
from testdriver.agentic import (
|
||||
DiscoveryRuntime, RealizationFailed, RecordedSelectorRuntime,
|
||||
)
|
||||
from testdriver.actions import SurfaceNotPermitted
|
||||
from testdriver.html import Document
|
||||
from lab.mutations import BY_ID, CATALOGUE
|
||||
from scenarios.browser_grant import baseline_recordings, build_agentic, lab_server
|
||||
|
||||
|
||||
def realize(*mutations: str, runtime=None):
|
||||
with lab_server(*mutations) as (app, tokens, base_url):
|
||||
world, driver, observer, asset, oracle = build_agentic(
|
||||
app, tokens, base_url, runtime
|
||||
)
|
||||
result = Runner(world, driver, observer, oracle).run(asset)
|
||||
surface = result.evidence.of_stratum(Stratum.SURFACE)[0]
|
||||
return result, surface.data
|
||||
|
||||
|
||||
def realized_ok(*mutations: str, runtime=None) -> bool:
|
||||
_, surface = realize(*mutations, runtime=runtime)
|
||||
return surface.get("raised") is None
|
||||
|
||||
|
||||
# --- the realization itself ----------------------------------------------
|
||||
|
||||
|
||||
def test_an_agent_realizes_the_semantic_action_from_intent():
|
||||
result, surface = realize()
|
||||
assert surface["raised"] is None
|
||||
assert result.verdict is Verdict.PASS
|
||||
assert surface["mechanics"]["runtime"] == "discovery-runtime"
|
||||
|
||||
|
||||
def test_the_runtime_is_never_handed_a_selector():
|
||||
"""Discovery must rest on meaning, not on identifiers it was given.
|
||||
|
||||
If the runtime consulted `data-td` it would be a recorded selector wearing a
|
||||
different hat, and H-001 would be measuring nothing.
|
||||
"""
|
||||
import inspect
|
||||
|
||||
from testdriver import agentic
|
||||
|
||||
source = inspect.getsource(agentic.DiscoveryRuntime)
|
||||
assert "data-td" not in source
|
||||
|
||||
|
||||
def test_deterministic_oracles_are_unchanged_by_agentic_realization():
|
||||
"""The agent may find the button. It may not decide whether that was right."""
|
||||
result, _ = realize()
|
||||
collectors = {
|
||||
obs.collector for obs in result.evidence.observations
|
||||
if obs.stratum is not Stratum.SURFACE
|
||||
}
|
||||
assert collectors == {"state-observer"}
|
||||
|
||||
|
||||
def test_a_defect_still_fails_under_agentic_realization():
|
||||
result, surface = realize("M17")
|
||||
assert surface["raised"] is None, "realization itself should succeed"
|
||||
assert result.verdict is Verdict.FAIL
|
||||
assert {j.assertion_id for j in result.judgments if j.verdict is Verdict.FAIL} == {
|
||||
"c-carol-denied", "i-enforcement-matches-record",
|
||||
}
|
||||
|
||||
|
||||
def test_the_agentic_asset_judges_fewer_claims_than_the_full_reference():
|
||||
"""A one-step scenario reaches fewer claims, and must not imply otherwise.
|
||||
|
||||
M15 passes here purely because this asset never revokes. That is correct and
|
||||
it is also exactly the kind of narrowing that silently overstates coverage if
|
||||
nobody writes it down.
|
||||
"""
|
||||
result, _ = realize("M15")
|
||||
judged = {j.assertion_id for j in result.judgments}
|
||||
assert "c-bob-revoked" not in judged
|
||||
assert result.verdict is Verdict.PASS
|
||||
|
||||
|
||||
# --- isolation and cost ---------------------------------------------------
|
||||
|
||||
|
||||
def test_each_actor_gets_its_own_session():
|
||||
"""A shared client is how isolation breaks in practice (F-0003)."""
|
||||
from testdriver.browser import BrowserDriver
|
||||
from testdriver.world import Actor
|
||||
|
||||
driver = BrowserDriver("http://127.0.0.1:1", {"a": "tok-a", "b": "tok-b"},
|
||||
DiscoveryRuntime(), "R")
|
||||
first = driver._session_for(Actor("a", "A"))
|
||||
second = driver._session_for(Actor("a", "A"))
|
||||
other = driver._session_for(Actor("b", "B"))
|
||||
assert first is not second, "sessions must not be cached across calls"
|
||||
assert first.token != other.token
|
||||
|
||||
|
||||
def test_cost_and_nondeterminism_are_recorded_from_the_first_run():
|
||||
"""Free to collect now, impossible to backfill later."""
|
||||
_, surface = realize()
|
||||
metrics = surface["mechanics"]["metrics"]
|
||||
for field in ("runtime", "wall_time_ms", "candidates_considered",
|
||||
"attempts", "retries", "tokens_in", "tokens_out", "model"):
|
||||
assert field in metrics
|
||||
assert metrics["candidates_considered"] > 0
|
||||
|
||||
|
||||
def test_runtime_identity_is_recorded_in_evidence():
|
||||
"""Which agent, which configuration — auditable after the fact."""
|
||||
_, surface = realize()
|
||||
assert surface["mechanics"]["metrics"]["runtime"] == "discovery-runtime"
|
||||
assert "rationale" in surface["mechanics"]
|
||||
|
||||
|
||||
# --- failing loudly -------------------------------------------------------
|
||||
|
||||
|
||||
def test_discovery_fails_loudly_when_the_control_is_absent():
|
||||
"""A missing control is not something to route around.
|
||||
|
||||
This is the guard that stops a removed authorization control from reading as
|
||||
a successful adaptation.
|
||||
"""
|
||||
page = Document.parse("<html><body><h1>Nothing here</h1></body></html>")
|
||||
with pytest.raises(RealizationFailed):
|
||||
DiscoveryRuntime().plan(page, "grant_access", {"subject_id": "bob"})
|
||||
|
||||
|
||||
def test_the_browser_action_may_not_be_performed_through_the_api():
|
||||
"""D-05: routing around the UI is surface substitution, not recovery."""
|
||||
action = SemanticAction(
|
||||
"grant_access", {}, permitted_surfaces=frozenset({"browser"})
|
||||
)
|
||||
with pytest.raises(SurfaceNotPermitted):
|
||||
action.check_surface("api")
|
||||
|
||||
|
||||
# --- H-001: the two arms --------------------------------------------------
|
||||
|
||||
MECHANICAL = [m for m in CATALOGUE if m.label == "MECHANICAL"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("mutation", MECHANICAL, ids=lambda m: m.id)
|
||||
def test_discovery_recovers_except_where_field_names_move(mutation):
|
||||
expected = mutation.id != "M22"
|
||||
assert realized_ok(mutation.id) is expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("mutation", MECHANICAL, ids=lambda m: m.id)
|
||||
def test_recorded_selectors_survive_exactly_while_test_ids_do(mutation):
|
||||
runtime = RecordedSelectorRuntime(baseline_recordings())
|
||||
assert realized_ok(mutation.id, runtime=runtime) is mutation.preserves_test_ids
|
||||
|
||||
|
||||
def test_the_control_arm_is_not_a_straw_man():
|
||||
"""Where test ids survive, the conventional approach loses nothing.
|
||||
|
||||
Asserting this protects H-001 from being flattered by a weak control.
|
||||
"""
|
||||
runtime = RecordedSelectorRuntime(baseline_recordings())
|
||||
preserved = [m for m in MECHANICAL if m.preserves_test_ids]
|
||||
assert all(realized_ok(m.id, runtime=runtime) for m in preserved)
|
||||
assert len(preserved) >= 9
|
||||
Loading…
Add table
Add a link
Reference in a new issue