fix: reject broken runtime launchers and preserve failed proof evidence
Some checks failed
Governed runtime contract / contract (push) Failing after 16s

Assistant: codex
Assistant-Model: gpt-6-astra
Assistant-Session: 01a0e387-534d-70e3-ad53-4ea05676db8c
This commit is contained in:
tegwick 2026-09-27 23:19:50 +02:00
parent 4e666d6fa3
commit 43e621439a
26 changed files with 1367 additions and 3 deletions

View file

@ -535,6 +535,34 @@ def test_profile_refusal_fails_terminally_without_legacy_fallback(
execute.assert_not_called()
def test_spend_refusal_closes_with_gateway_cleanup_evidence(tmp_path: Path) -> None:
from rein_aharness.glas_execution import GlasSpendError
_repo, run, client = _profiled_case(tmp_path)
client.fail.return_value = OpsRun(
id=run.id, activity_definition_id="def", idempotency_key="k",
target_repo=run.target_repo, title=run.title, description="", state="failed",
)
refusal = GlasSpendError(
"spend accounting refused: execution accounting requires reconciliation",
execution_evidence={
"outcome": "failed", "failure_stage": "execution",
"session_cleanup": "succeeded", "sandbox_destroy": "succeeded",
"error": "private response", "provider_response": "private payload",
},
)
with patch("rein_aharness.claim_loop.execute_profiled_run", side_effect=refusal):
result = process_one(client)
assert result.ok is False
assert client.fail.call_args.kwargs["reopen"] is False
evidence = client.fail.call_args.kwargs["result"]["execution_evidence"]
assert evidence == {
"outcome": "failed", "failure_stage": "execution",
"session_cleanup": "succeeded", "sandbox_destroy": "succeeded",
}
client.complete.assert_not_called()
def test_profiled_signal_cancellation_releases_lock_and_durably_fails(
tmp_path: Path,
) -> None:

View file

@ -0,0 +1,53 @@
"""Frozen action packet safety; requires the governed secrets-engine sibling."""
import importlib.util
from pathlib import Path
import shutil
import pytest
pytest.importorskip("secrets_engine")
ROOT = Path(__file__).resolve().parents[1]
SOURCE = ROOT.parent / "secrets-engine/docs/proposals/glas-metered-tool-renewal-20260927"
spec = importlib.util.spec_from_file_location("metered_native", ROOT / "scripts/metered-native-owner.py")
native = importlib.util.module_from_spec(spec)
spec.loader.exec_module(native)
@pytest.fixture
def bundle(tmp_path):
path = tmp_path / "bundle"
shutil.copytree(SOURCE, path)
return path
def test_all_six_action_bindings_match(bundle, tmp_path):
private = tmp_path / "private"
private.mkdir()
configs, entries = native.prepare_configs(bundle, private)
assert len(entries) == 6
for (lane, action), entry in entries.items():
assert entry.approval["authorization_id"] == native.IDS[lane][action]
assert configs["exec"].clock_trust_file is None # validation is backend-free
@pytest.mark.parametrize("lane", [native.PROVIDER, native.WORKER])
def test_recipient_drift_refused_before_auth(bundle, tmp_path, lane):
path = bundle / "catalog" / (lane + ".yaml")
path.write_text(path.read_text().replace("token_max_ttl: 15m", "token_max_ttl: 16m"))
private = tmp_path / "private"
private.mkdir()
with pytest.raises(ValueError, match="frozen_action_request_drift"):
native.prepare_configs(bundle, private)
def test_changed_packet_refused(bundle, tmp_path):
path = bundle / "native-pdp-inputs.json"
path.write_bytes(path.read_bytes() + b"\n")
with pytest.raises(ValueError, match="frozen_packet_drift"):
native.prepare_configs(bundle, tmp_path)
def test_execution_never_replays_prior_attempt(bundle):
(bundle / "execution.json").write_text('{"phase":"one_exec_started"}')
with pytest.raises(ValueError, match="prior_attempt_requires_reconciliation"):
native.execute(bundle)

View file

@ -0,0 +1,46 @@
"""The one-row repair cannot widen or synthesize mutation authority."""
import importlib.util
from pathlib import Path
from types import SimpleNamespace
from copy import deepcopy
import pytest
spec = importlib.util.spec_from_file_location("repair", Path(__file__).resolve().parents[1] / "scripts/reconcile-metered-queue-grant.py")
repair = importlib.util.module_from_spec(spec)
spec.loader.exec_module(repair)
IMAGE = "sha256:713bddad10a41950f446100a8b370fca8ccdd0b8969cccae751e3870c9c63ccd"
@pytest.fixture
def records():
action = dict(repository_grant=deepcopy(repair.GRANT), target_repo="hfact-glas-proof", harness_profile_ref="harness.agent-dev-local@1.1.1", labels=["hfact-metered"], description="bounded task", task_template="proof")
definition = SimpleNamespace(id=repair.DEFINITION, enabled=False, version=1, rules_json=[dict(id="execute-hfact-glas-metered-proof", condition="True", action=action)])
row = SimpleNamespace(id=repair.RUN, activity_definition_id=repair.DEFINITION, state="open", attempt=0, claim_owner=None, lease_until=None, target_repo="hfact-glas-proof", harness_profile_ref="harness.agent-dev-local@1.1.1", triggering_event_id=repair.TRIGGER, source_type="rule", source_id="execute-hfact-glas-metered-proof", repository_grant=None, description="bounded task", title="proof", labels=["hfact-metered"])
return definition, row
def test_copies_only_typed_existing_authority(records):
definition, row = records
assert repair.validate(definition, row, IMAGE) is definition.rules_json[0]["action"]["repository_grant"]
assert row.repository_grant is None
@pytest.mark.parametrize("field,value", [("state", "claimed"), ("attempt", 1), ("claim_owner", "worker"), ("repository_grant", repair.GRANT), ("triggering_event_id", "another-trigger"), ("description", "changed task")])
def test_changed_or_used_row_refused(records, field, value):
definition, row = records
setattr(row, field, value)
with pytest.raises(ValueError):
repair.validate(definition, row, IMAGE)
def test_changed_definition_cannot_grant_more(records):
definition, row = records
definition.rules_json[0]["action"]["repository_grant"]["allowed_paths"] = ["**"]
with pytest.raises(ValueError, match="typed_authority_drift"):
repair.validate(definition, row, IMAGE)
def test_stale_worker_blocks_repair(records):
with pytest.raises(ValueError, match="grant_aware_worker_required"):
repair.validate(*records, "sha256:old")

View file

@ -42,6 +42,10 @@ def prepared(ledger, tmp_path, monkeypatch):
runtime = tmp_path / "runtime"
(runtime / "bin").mkdir(parents=True)
(runtime / "bin/python3").write_bytes(b"not executed; fixture runtime structure")
for name in ("rein-aharness", "glas-harness"):
executable = runtime / "bin" / name
executable.write_text("#!/opt/sandboxer/runtime/bin/python3\n# fixture only\n")
executable.chmod(0o755)
(runtime / "pyvenv.cfg").write_text("fixture only")
messages = MessagesPolicy("fixture:upper-rate", profile.model.model, 1000, 1000, 1000, 1000)
data = {"version": "1", "authority_ref": policy.authority_ref,
@ -62,7 +66,8 @@ def test_prepare_pins_without_key_or_claim(prepared, monkeypatch):
@pytest.mark.parametrize("case", ["authority", "policy_digest", "model", "version", "unknown_field",
"duplicate", "public", "runtime_changed", "schema_missing"])
"duplicate", "public", "runtime_changed", "schema_missing",
"build_host_launcher", "missing_launcher", "nonexecutable_launcher"])
def test_bad_bootstrap_refuses_before_claim(prepared, monkeypatch, capsys, case):
path, config, data, runtime = prepared
if case == "authority": data["authority_ref"] = "unrelated"
@ -71,6 +76,16 @@ def test_bad_bootstrap_refuses_before_claim(prepared, monkeypatch, capsys, case)
elif case == "version": data["version"] = "2"
elif case == "unknown_field": data["upstream_url"] = "https://not-admitted.invalid"
elif case == "runtime_changed": (runtime / "added-file").write_text("changed")
elif case in {"build_host_launcher", "missing_launcher", "nonexecutable_launcher"}:
executable = runtime / "bin/rein-aharness"
if case == "build_host_launcher":
executable.write_text("#!/bin/sh\nexec /old-build/bin/python3\n")
elif case == "missing_launcher":
executable.unlink()
else:
executable.chmod(0o644)
# Even a correctly pinned malformed runtime must fail before claim.
data["runtime"]["sha256"] = runtime_digest(runtime)
elif case == "schema_missing":
import sqlite3
with sqlite3.connect(config.spend_ledger_path) as db: db.execute("DROP TABLE request_routes")

View file

@ -419,6 +419,29 @@ def test_uncertain_gateway_never_imports_and_blocks_next_claim(dispatch_case, fa
store.preflight()
def test_accounting_refusal_preserves_stage_and_cleanup_without_raw_error(dispatch_case):
from rein_aharness.glas_execution import GlasSpendError, execute_profiled_run
store, config, _catalog, transfer = dispatch_case
result = success(store, run(), None)
result["ok"] = False
result["evidence"].update(
outcome="failed", failure_stage="execution", sandbox_id="fixture-sandbox",
error="private provider response", provider_response="private payload",
)
with pytest.raises(GlasSpendError) as caught:
execute_profiled_run(run(), config, gateway=lambda *a, **k: result, report_to_hub=False)
evidence = caught.value.execution_evidence
assert evidence["failure_stage"] == "execution"
assert evidence["sandbox_destroy"] == evidence["session_cleanup"] == "succeeded"
assert "private" not in json.dumps(evidence)
assert "error" not in evidence and "cost_usd" not in evidence
transfer.import_after_teardown.assert_not_called()
assert store.status()["reservations"][0]["state"] == "held"
with pytest.raises(SpendAdmissionError):
store.preflight()
def test_cli_reconciliation_requires_termination_attestation(ledger, capsys):
from rein_aharness.cli import main