Advance supervised agent records and close verified Secret annotation guard
All checks were successful
CI Smoke / host-smoke (push) Successful in 0s
CI Smoke / container-smoke (push) Successful in 3s
Python Tests / pytest (push) Successful in 25s

This commit is contained in:
codex 2026-09-28 18:15:27 +02:00
parent b2f6713721
commit db91818e84
44 changed files with 6868 additions and 54 deletions

View file

@ -0,0 +1,84 @@
#!/usr/bin/env python3
"""CUST-WP-0073-T02: exercise existing sand-boxer isolation without a model call.
Run with ~/glas-harness/.venv/bin/python. This does not switch the current
interactive agent into the sandbox or certify a complete agent runtime.
"""
import json
import tempfile
from datetime import datetime, timezone
from pathlib import Path
from sandboxer.core.manager import SandboxManager
from sandboxer.lifecycle.store import SandboxStore
from sandboxer.models import Consumer, SandboxCreateRequest, SandboxExecRequest
from sandboxer.payments.credits import CreditsStore
from sandboxer.snapshots.store import SnapshotStore
PROBE = r'''
import json, os, socket
from pathlib import Path
paths = ['/home/worsch', '/home/tegwick', '/root', '/etc/rancher/k3s',
'/run/docker.sock', '/var/run/docker.sock', '/run/containerd',
'/run/user/1000', '/mnt/c']
checks = {'admin_paths_absent': all(not Path(p).exists() for p in paths),
'admin_environment_absent': not any(os.environ.get(k) for k in
['SSH_AUTH_SOCK', 'KUBECONFIG', 'BAO_TOKEN', 'VAULT_TOKEN']),
'only_loopback_interface': socket.if_nameindex() == [(1, 'lo')],
'approved_observation_readable': Path('observation.txt').read_text() == 'synthetic observation\n'}
Path('proposal.txt').write_text('synthetic privileged action proposal; not executed\n')
checks['proposal_preparation_works'] = Path('proposal.txt').is_file()
print(json.dumps(checks))
'''
def main():
report = {"captured_at": datetime.now(timezone.utc).isoformat(),
"workplan_task": "CUST-WP-0073-T02", "profile": "profile.bwrap-local",
"model_called": False, "interactive_agent_migrated": False}
with tempfile.TemporaryDirectory(prefix="cust-supervised-proof-") as temp:
root = Path(temp)
source = root / "source"
source.mkdir()
(source / "observation.txt").write_text("synthetic observation\n")
manager = SandboxManager(
store=SandboxStore(path=root / "sandboxes.json"),
credits=CreditsStore(path=root / "credits.json"),
snapshots=SnapshotStore(path=root / "snapshots.json"),
)
consumer = Consumer(actor="agt", project="the-custodian",
run_id="cust-wp-0073-supervised-proof")
status = manager.create(SandboxCreateRequest(
profile="profile.bwrap-local", inputs={"repo": str(source)},
consumer=consumer, ttl="5m",
))
report["sandbox_id"] = status.sandbox_id
try:
result = manager.execute(status.sandbox_id, SandboxExecRequest(
command=["/usr/bin/python3", "-c", PROBE], consumer=consumer,
timeout_seconds=15,
))
if result.exit_code or result.timed_out or result.output_truncated:
raise RuntimeError("sandbox probe failed; child output suppressed")
report["checks"] = json.loads(result.stdout)
wrong = Consumer(actor="agt", project="the-custodian", run_id="wrong-run")
try:
manager.execute(status.sandbox_id, SandboxExecRequest(
command=["/bin/true"], consumer=wrong, timeout_seconds=5))
except (ValueError, PermissionError):
report["checks"]["wrong_consumer_denied"] = True
else:
report["checks"]["wrong_consumer_denied"] = False
finally:
destroyed = manager.destroy(status.sandbox_id)
report["checks"]["workspace_removed"] = not Path(status.reachability.workspace_dir).exists()
report["checks"]["destroyed"] = destroyed.state.value == "destroyed"
report["checks"]["host_source_unchanged"] = not (source / "proposal.txt").exists()
report["passed"] = all(report["checks"].values())
print(json.dumps(report, indent=2))
return 0 if report["passed"] else 1
if __name__ == "__main__":
raise SystemExit(main())

View file

@ -0,0 +1,80 @@
#!/usr/bin/env python3
"""Summarize per-agent supervised proposals; never grant or execute authority."""
import argparse
import json
from collections import Counter
from pathlib import Path
def summarize(record):
if record.get("mode") not in {"supervised", "autopilot"}:
raise ValueError("invalid agent mode")
if not record.get("agent_id") or not record.get("scope"):
raise ValueError("agent_id and scope required")
seen = set()
counts = Counter()
for proposal in record["proposals"]:
identity = proposal["proposal_id"]
if identity in seen:
raise ValueError("duplicate original proposal; revisions are not new trials")
seen.add(identity)
decision = proposal["disposition"]
outcome = proposal["outcome"]
if decision not in {"pending", "withdrawn", "accepted_unchanged", "accepted_revised", "rejected", "unscored"}:
raise ValueError("invalid disposition")
if outcome not in {"not_executed", "unverified", "succeeded", "failed"}:
raise ValueError("invalid outcome")
counts[decision] += 1
counts["total"] += 1
counts["outcome_" + outcome] += 1
if proposal.get("refinement_or_rescue") is True:
counts["refinement_or_rescue"] += 1
if decision == "unscored":
# Historic/broad conversation approval lacks an exact submitted
# revision. Preserve it without fabricating promotion evidence.
continue
if decision in {"accepted_unchanged", "accepted_revised", "rejected"}:
counts["adjudicated"] += 1
if outcome != "not_executed" and decision not in {"accepted_unchanged", "accepted_revised"}:
raise ValueError("executed trial requires an accepted proposal")
if decision in {"accepted_unchanged", "accepted_revised"}:
if not proposal.get("approval_ref") or not proposal.get("original_digest") or not proposal.get("approved_digest"):
raise ValueError("scored acceptance requires exact proposal and approval references")
equal = proposal["original_digest"] == proposal["approved_digest"]
if equal != (decision == "accepted_unchanged"):
raise ValueError("disposition disagrees with approved revision")
if outcome == "unverified":
counts["unverified"] += 1
if outcome in {"succeeded", "failed"}:
if not proposal.get("verification_ref") or not proposal.get("executed_digest"):
raise ValueError("verified execution requires receipt and exact executed revision")
if proposal["executed_digest"] != proposal["approved_digest"]:
raise ValueError("execution differs from approved revision")
if type(proposal.get("refinement_or_rescue")) is not bool:
raise ValueError("execution refinement/rescue must be explicit")
counts["verified_executions"] += 1
if outcome == "succeeded" and decision == "accepted_unchanged" and not proposal["refinement_or_rescue"]:
counts["unchanged_successes"] += 1
def rate(numerator, denominator):
return counts[numerator] / counts[denominator] if counts[denominator] else None
return {"agent_id": record["agent_id"], "scope": record["scope"],
"mode": record["mode"], "counts": dict(counts),
"unchanged_acceptance_rate": rate("accepted_unchanged", "adjudicated"),
"unchanged_execution_success_rate": rate("unchanged_successes", "verified_executions"),
"authority_granted": False,
"note": "Descriptive evidence only; null rates mean no eligible sample. Never promotes an agent."}
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("record", type=Path)
args = parser.parse_args()
try:
report = summarize(json.loads(args.record.read_text()))
except (ValueError, KeyError, TypeError, OSError):
parser.exit(2, "Invalid supervision record; no report produced.\n")
print(json.dumps(report, indent=2))
if __name__ == "__main__":
main()