fix: reject broken runtime launchers and preserve failed proof evidence
Some checks failed
Governed runtime contract / contract (push) Failing after 16s
Some checks failed
Governed runtime contract / contract (push) Failing after 16s
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e387-534d-70e3-ad53-4ea05676db8c
This commit is contained in:
parent
4e666d6fa3
commit
43e621439a
26 changed files with 1367 additions and 3 deletions
|
|
@ -535,6 +535,34 @@ def test_profile_refusal_fails_terminally_without_legacy_fallback(
|
|||
execute.assert_not_called()
|
||||
|
||||
|
||||
def test_spend_refusal_closes_with_gateway_cleanup_evidence(tmp_path: Path) -> None:
|
||||
from rein_aharness.glas_execution import GlasSpendError
|
||||
|
||||
_repo, run, client = _profiled_case(tmp_path)
|
||||
client.fail.return_value = OpsRun(
|
||||
id=run.id, activity_definition_id="def", idempotency_key="k",
|
||||
target_repo=run.target_repo, title=run.title, description="", state="failed",
|
||||
)
|
||||
refusal = GlasSpendError(
|
||||
"spend accounting refused: execution accounting requires reconciliation",
|
||||
execution_evidence={
|
||||
"outcome": "failed", "failure_stage": "execution",
|
||||
"session_cleanup": "succeeded", "sandbox_destroy": "succeeded",
|
||||
"error": "private response", "provider_response": "private payload",
|
||||
},
|
||||
)
|
||||
with patch("rein_aharness.claim_loop.execute_profiled_run", side_effect=refusal):
|
||||
result = process_one(client)
|
||||
assert result.ok is False
|
||||
assert client.fail.call_args.kwargs["reopen"] is False
|
||||
evidence = client.fail.call_args.kwargs["result"]["execution_evidence"]
|
||||
assert evidence == {
|
||||
"outcome": "failed", "failure_stage": "execution",
|
||||
"session_cleanup": "succeeded", "sandbox_destroy": "succeeded",
|
||||
}
|
||||
client.complete.assert_not_called()
|
||||
|
||||
|
||||
def test_profiled_signal_cancellation_releases_lock_and_durably_fails(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
|
|
|
|||
53
tests/test_metered_native_procedure.py
Normal file
53
tests/test_metered_native_procedure.py
Normal file
|
|
@ -0,0 +1,53 @@
|
|||
"""Frozen action packet safety; requires the governed secrets-engine sibling."""
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
|
||||
import pytest
|
||||
|
||||
pytest.importorskip("secrets_engine")
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
SOURCE = ROOT.parent / "secrets-engine/docs/proposals/glas-metered-tool-renewal-20260927"
|
||||
spec = importlib.util.spec_from_file_location("metered_native", ROOT / "scripts/metered-native-owner.py")
|
||||
native = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(native)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def bundle(tmp_path):
|
||||
path = tmp_path / "bundle"
|
||||
shutil.copytree(SOURCE, path)
|
||||
return path
|
||||
|
||||
|
||||
def test_all_six_action_bindings_match(bundle, tmp_path):
|
||||
private = tmp_path / "private"
|
||||
private.mkdir()
|
||||
configs, entries = native.prepare_configs(bundle, private)
|
||||
assert len(entries) == 6
|
||||
for (lane, action), entry in entries.items():
|
||||
assert entry.approval["authorization_id"] == native.IDS[lane][action]
|
||||
assert configs["exec"].clock_trust_file is None # validation is backend-free
|
||||
|
||||
|
||||
@pytest.mark.parametrize("lane", [native.PROVIDER, native.WORKER])
|
||||
def test_recipient_drift_refused_before_auth(bundle, tmp_path, lane):
|
||||
path = bundle / "catalog" / (lane + ".yaml")
|
||||
path.write_text(path.read_text().replace("token_max_ttl: 15m", "token_max_ttl: 16m"))
|
||||
private = tmp_path / "private"
|
||||
private.mkdir()
|
||||
with pytest.raises(ValueError, match="frozen_action_request_drift"):
|
||||
native.prepare_configs(bundle, private)
|
||||
|
||||
|
||||
def test_changed_packet_refused(bundle, tmp_path):
|
||||
path = bundle / "native-pdp-inputs.json"
|
||||
path.write_bytes(path.read_bytes() + b"\n")
|
||||
with pytest.raises(ValueError, match="frozen_packet_drift"):
|
||||
native.prepare_configs(bundle, tmp_path)
|
||||
|
||||
|
||||
def test_execution_never_replays_prior_attempt(bundle):
|
||||
(bundle / "execution.json").write_text('{"phase":"one_exec_started"}')
|
||||
with pytest.raises(ValueError, match="prior_attempt_requires_reconciliation"):
|
||||
native.execute(bundle)
|
||||
46
tests/test_metered_queue_reconciliation.py
Normal file
46
tests/test_metered_queue_reconciliation.py
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
"""The one-row repair cannot widen or synthesize mutation authority."""
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
from copy import deepcopy
|
||||
|
||||
import pytest
|
||||
|
||||
spec = importlib.util.spec_from_file_location("repair", Path(__file__).resolve().parents[1] / "scripts/reconcile-metered-queue-grant.py")
|
||||
repair = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(repair)
|
||||
IMAGE = "sha256:713bddad10a41950f446100a8b370fca8ccdd0b8969cccae751e3870c9c63ccd"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def records():
|
||||
action = dict(repository_grant=deepcopy(repair.GRANT), target_repo="hfact-glas-proof", harness_profile_ref="harness.agent-dev-local@1.1.1", labels=["hfact-metered"], description="bounded task", task_template="proof")
|
||||
definition = SimpleNamespace(id=repair.DEFINITION, enabled=False, version=1, rules_json=[dict(id="execute-hfact-glas-metered-proof", condition="True", action=action)])
|
||||
row = SimpleNamespace(id=repair.RUN, activity_definition_id=repair.DEFINITION, state="open", attempt=0, claim_owner=None, lease_until=None, target_repo="hfact-glas-proof", harness_profile_ref="harness.agent-dev-local@1.1.1", triggering_event_id=repair.TRIGGER, source_type="rule", source_id="execute-hfact-glas-metered-proof", repository_grant=None, description="bounded task", title="proof", labels=["hfact-metered"])
|
||||
return definition, row
|
||||
|
||||
|
||||
def test_copies_only_typed_existing_authority(records):
|
||||
definition, row = records
|
||||
assert repair.validate(definition, row, IMAGE) is definition.rules_json[0]["action"]["repository_grant"]
|
||||
assert row.repository_grant is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field,value", [("state", "claimed"), ("attempt", 1), ("claim_owner", "worker"), ("repository_grant", repair.GRANT), ("triggering_event_id", "another-trigger"), ("description", "changed task")])
|
||||
def test_changed_or_used_row_refused(records, field, value):
|
||||
definition, row = records
|
||||
setattr(row, field, value)
|
||||
with pytest.raises(ValueError):
|
||||
repair.validate(definition, row, IMAGE)
|
||||
|
||||
|
||||
def test_changed_definition_cannot_grant_more(records):
|
||||
definition, row = records
|
||||
definition.rules_json[0]["action"]["repository_grant"]["allowed_paths"] = ["**"]
|
||||
with pytest.raises(ValueError, match="typed_authority_drift"):
|
||||
repair.validate(definition, row, IMAGE)
|
||||
|
||||
|
||||
def test_stale_worker_blocks_repair(records):
|
||||
with pytest.raises(ValueError, match="grant_aware_worker_required"):
|
||||
repair.validate(*records, "sha256:old")
|
||||
|
|
@ -42,6 +42,10 @@ def prepared(ledger, tmp_path, monkeypatch):
|
|||
runtime = tmp_path / "runtime"
|
||||
(runtime / "bin").mkdir(parents=True)
|
||||
(runtime / "bin/python3").write_bytes(b"not executed; fixture runtime structure")
|
||||
for name in ("rein-aharness", "glas-harness"):
|
||||
executable = runtime / "bin" / name
|
||||
executable.write_text("#!/opt/sandboxer/runtime/bin/python3\n# fixture only\n")
|
||||
executable.chmod(0o755)
|
||||
(runtime / "pyvenv.cfg").write_text("fixture only")
|
||||
messages = MessagesPolicy("fixture:upper-rate", profile.model.model, 1000, 1000, 1000, 1000)
|
||||
data = {"version": "1", "authority_ref": policy.authority_ref,
|
||||
|
|
@ -62,7 +66,8 @@ def test_prepare_pins_without_key_or_claim(prepared, monkeypatch):
|
|||
|
||||
|
||||
@pytest.mark.parametrize("case", ["authority", "policy_digest", "model", "version", "unknown_field",
|
||||
"duplicate", "public", "runtime_changed", "schema_missing"])
|
||||
"duplicate", "public", "runtime_changed", "schema_missing",
|
||||
"build_host_launcher", "missing_launcher", "nonexecutable_launcher"])
|
||||
def test_bad_bootstrap_refuses_before_claim(prepared, monkeypatch, capsys, case):
|
||||
path, config, data, runtime = prepared
|
||||
if case == "authority": data["authority_ref"] = "unrelated"
|
||||
|
|
@ -71,6 +76,16 @@ def test_bad_bootstrap_refuses_before_claim(prepared, monkeypatch, capsys, case)
|
|||
elif case == "version": data["version"] = "2"
|
||||
elif case == "unknown_field": data["upstream_url"] = "https://not-admitted.invalid"
|
||||
elif case == "runtime_changed": (runtime / "added-file").write_text("changed")
|
||||
elif case in {"build_host_launcher", "missing_launcher", "nonexecutable_launcher"}:
|
||||
executable = runtime / "bin/rein-aharness"
|
||||
if case == "build_host_launcher":
|
||||
executable.write_text("#!/bin/sh\nexec /old-build/bin/python3\n")
|
||||
elif case == "missing_launcher":
|
||||
executable.unlink()
|
||||
else:
|
||||
executable.chmod(0o644)
|
||||
# Even a correctly pinned malformed runtime must fail before claim.
|
||||
data["runtime"]["sha256"] = runtime_digest(runtime)
|
||||
elif case == "schema_missing":
|
||||
import sqlite3
|
||||
with sqlite3.connect(config.spend_ledger_path) as db: db.execute("DROP TABLE request_routes")
|
||||
|
|
|
|||
|
|
@ -419,6 +419,29 @@ def test_uncertain_gateway_never_imports_and_blocks_next_claim(dispatch_case, fa
|
|||
store.preflight()
|
||||
|
||||
|
||||
def test_accounting_refusal_preserves_stage_and_cleanup_without_raw_error(dispatch_case):
|
||||
from rein_aharness.glas_execution import GlasSpendError, execute_profiled_run
|
||||
|
||||
store, config, _catalog, transfer = dispatch_case
|
||||
result = success(store, run(), None)
|
||||
result["ok"] = False
|
||||
result["evidence"].update(
|
||||
outcome="failed", failure_stage="execution", sandbox_id="fixture-sandbox",
|
||||
error="private provider response", provider_response="private payload",
|
||||
)
|
||||
with pytest.raises(GlasSpendError) as caught:
|
||||
execute_profiled_run(run(), config, gateway=lambda *a, **k: result, report_to_hub=False)
|
||||
evidence = caught.value.execution_evidence
|
||||
assert evidence["failure_stage"] == "execution"
|
||||
assert evidence["sandbox_destroy"] == evidence["session_cleanup"] == "succeeded"
|
||||
assert "private" not in json.dumps(evidence)
|
||||
assert "error" not in evidence and "cost_usd" not in evidence
|
||||
transfer.import_after_teardown.assert_not_called()
|
||||
assert store.status()["reservations"][0]["state"] == "held"
|
||||
with pytest.raises(SpendAdmissionError):
|
||||
store.preflight()
|
||||
|
||||
|
||||
def test_cli_reconciliation_requires_termination_attestation(ledger, capsys):
|
||||
from rein_aharness.cli import main
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue