From 73aaa4bcd49585cd8e33aac259c90655fecb5aae Mon Sep 17 00:00:00 2001 From: tegwick Date: Sun, 27 Sep 2026 17:46:34 +0200 Subject: [PATCH] Close local runtime loose ends and record external blockers Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e387-534d-70e3-ad53-4ea05676db8c --- .forgejo/workflows/runtime-contract.yaml | 2 +- WORK-RECORDS.md | 6 +- .../2026-09-27-sonnet5-runtime-proof.json | 242 ++++++++++++++ ...2026-09-27-sonnet5-tool-session-proof.json | 310 ++++++++++++++++++ docs/owner-bootstrap.md | 22 ++ scripts/prove-metered-runtime.py | 99 +++++- .../REINAH-WP-0001-harness-foundation.md | 2 +- ...INAH-WP-0003-governed-runtime-integrity.md | 93 +++++- 8 files changed, 752 insertions(+), 24 deletions(-) create mode 100644 docs/evidence/2026-09-27-sonnet5-runtime-proof.json create mode 100644 docs/evidence/2026-09-27-sonnet5-tool-session-proof.json diff --git a/.forgejo/workflows/runtime-contract.yaml b/.forgejo/workflows/runtime-contract.yaml index affe76d..4ba88fb 100644 --- a/.forgejo/workflows/runtime-contract.yaml +++ b/.forgejo/workflows/runtime-contract.yaml @@ -70,4 +70,4 @@ jobs: "${uv_bin}" sync \ --python "${python_bin}" \ --frozen --no-editable --extra glas --extra llm --extra dev - ./scripts/verify-runtime-contracts.sh + make contract-test recovery-test diff --git a/WORK-RECORDS.md b/WORK-RECORDS.md index 65ab2e4..f500d27 100644 --- a/WORK-RECORDS.md +++ b/WORK-RECORDS.md @@ -9,9 +9,9 @@ | Kind | ID | Status | Lane | Source | | --- | --- | --- | --- | --- | | workplan | HARNESS-WP-ADHOC-2026-09-04 | finished | — | workplans/ADHOC-2026-09-04.md | -| workplan | REINAH-WP-0001 | done | — | workplans/REINAH-WP-0001-harness-foundation.md | +| workplan | REINAH-WP-0001 | finished | — | workplans/REINAH-WP-0001-harness-foundation.md | | workplan | REINAH-WP-0002 | finished | — | workplans/REINAH-WP-0002-rename-and-glas-harness-alignment.md | -| workplan | REINAH-WP-0003 | active | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md | +| workplan | REINAH-WP-0003 | blocked | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md | | workplan | REINAH-WP-0004 | finished | — | workplans/REINAH-WP-0004-statehub-bootstrap.md | | workplan | REINAH-WP-0005 | finished | — | workplans/REINAH-WP-0005-ops-run-claim-loop.md | | workplan | REINAH-WP-0006 | finished | — | workplans/REINAH-WP-0006-binky-weekly-review-approach.md | @@ -32,7 +32,7 @@ | task | REINAH-WP-0003-T02 | done | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md | | task | REINAH-WP-0003-T03 | done | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md | | task | REINAH-WP-0003-T04 | wait | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md | -| task | REINAH-WP-0003-T05 | progress | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md | +| task | REINAH-WP-0003-T05 | wait | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md | | task | REINAH-WP-0003-T06 | wait | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md | | task | REINAH-WP-0004-T01 | done | — | workplans/REINAH-WP-0004-statehub-bootstrap.md | | task | REINAH-WP-0004-T02 | done | — | workplans/REINAH-WP-0004-statehub-bootstrap.md | diff --git a/docs/evidence/2026-09-27-sonnet5-runtime-proof.json b/docs/evidence/2026-09-27-sonnet5-runtime-proof.json new file mode 100644 index 0000000..d03f76f --- /dev/null +++ b/docs/evidence/2026-09-27-sonnet5-runtime-proof.json @@ -0,0 +1,242 @@ +{ + "scope": "standalone installed packages and real CLI/bwrap; synthetic key/provider/empty queue", + "ok": true, + "runtime_sha256": "b6e4e8a429393d68831c996a65c0489664205df969ac9882e581983ec2da4969", + "proof_script_sha256": "a3cde37c6b1075caaa092248569d9b7a3e37bde467648686d70ac77ef86c89d8", + "claude_sha256": "19842705e989393fce936804df6d2ab034860e24b8f8880357981d87ffd83fac", + "imports": { + "rein_aharness": "lib/python3.12/site-packages/rein_aharness/__init__.py", + "llm_connect": "lib/python3.12/site-packages/llm_connect/__init__.py", + "glas_harness": "lib/python3.12/site-packages/glas_harness/__init__.py", + "sandboxer": "lib/python3.12/site-packages/sandboxer/__init__.py" + }, + "packaged_definitions": true, + "positive": { + "native_threshold_usd": 1.0, + "probe": { + "runtime_readonly": true, + "python_prefix": "/opt/sandboxer/runtime", + "cli_version": "2.1.266 (Claude Code)", + "private_absent": true, + "source_absent": true, + "proxy_absent": true, + "interfaces": [ + "lo" + ] + }, + "provider_requests": 1, + "cli_exit_code": 0, + "cli_is_error": false, + "cli_estimated_usd": 0.00030000000000000003, + "request_reservations": [ + { + "receipt": "5519ba2f-fe1f-4928-8afb-fb765acc1bdc", + "run_id": "fixture-run", + "policy_sha256": "c72c3a7c69e5224c63beb60f699bbc0c28d93d4b300604cae4208e761a1b4825", + "lease_id": "d40ebad662643b376f406fb07bf37671cb99497cc23755e3e8f97cda53b37a29", + "liability_microusd": 4640000, + "state": "charged", + "observed_microusd": 500, + "created_at": "2026-09-27T14:55:51.865416+00:00" + } + ], + "route_revoked": true, + "workspace_removed": true, + "bootstrap": { + "ok": true, + "claimed": false, + "empty": true, + "run_id": "", + "ops_state": null + }, + "profile_ref": "harness.agent-dev-local@1.1.1", + "model": "claude-sonnet-5", + "request_shapes": [ + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "adaptive", + "manual_thinking_budget": false, + "sampling_parameters": {}, + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24" + ] + } + ], + "policy_context_tokens": 1000000, + "admission_requests": [ + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "disabled", + "top_level_fields": [ + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24", + "structured-outputs-2025-12-15" + ], + "message_roles": [ + "user" + ], + "output_config_keys": [ + "effort", + "format" + ], + "output_effort": "high", + "refusal": "beta feature not admitted" + }, + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "adaptive", + "top_level_fields": [ + "context_management", + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24" + ], + "message_roles": [ + "user" + ], + "output_config_keys": [ + "effort" + ], + "output_effort": "high" + } + ] + }, + "refused": { + "native_threshold_usd": 0.01, + "probe": { + "runtime_readonly": true, + "python_prefix": "/opt/sandboxer/runtime", + "cli_version": "2.1.266 (Claude Code)", + "private_absent": true, + "source_absent": true, + "proxy_absent": true, + "interfaces": [ + "lo" + ] + }, + "provider_requests": 0, + "cli_exit_code": 1, + "cli_is_error": true, + "cli_estimated_usd": 0, + "request_reservations": [], + "route_revoked": true, + "workspace_removed": true, + "bootstrap": null, + "profile_ref": "harness.agent-dev-local@1.1.1", + "model": "claude-sonnet-5", + "request_shapes": [], + "policy_context_tokens": 1000000, + "admission_requests": [ + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "disabled", + "top_level_fields": [ + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24", + "structured-outputs-2025-12-15" + ], + "message_roles": [ + "user" + ], + "output_config_keys": [ + "effort", + "format" + ], + "output_effort": "high", + "refusal": "beta feature not admitted" + }, + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "adaptive", + "top_level_fields": [ + "context_management", + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24" + ], + "message_roles": [ + "user" + ], + "output_config_keys": [ + "effort" + ], + "output_effort": "high" + } + ] + }, + "artifact_unchanged": true, + "live_factory_attempts": 0 +} diff --git a/docs/evidence/2026-09-27-sonnet5-tool-session-proof.json b/docs/evidence/2026-09-27-sonnet5-tool-session-proof.json new file mode 100644 index 0000000..4a6754e --- /dev/null +++ b/docs/evidence/2026-09-27-sonnet5-tool-session-proof.json @@ -0,0 +1,310 @@ +{ + "scope": "standalone installed packages and real CLI/bwrap; synthetic key/provider/empty queue", + "ok": true, + "runtime_sha256": "b6e4e8a429393d68831c996a65c0489664205df969ac9882e581983ec2da4969", + "proof_script_sha256": "984c062d3e4c36671065803b1e2706f81388f31b97a8a5af9d488a31e96ca5b2", + "claude_sha256": "19842705e989393fce936804df6d2ab034860e24b8f8880357981d87ffd83fac", + "imports": { + "rein_aharness": "lib/python3.12/site-packages/rein_aharness/__init__.py", + "llm_connect": "lib/python3.12/site-packages/llm_connect/__init__.py", + "glas_harness": "lib/python3.12/site-packages/glas_harness/__init__.py", + "sandboxer": "lib/python3.12/site-packages/sandboxer/__init__.py" + }, + "packaged_definitions": true, + "positive": { + "native_threshold_usd": 1.0, + "probe": { + "runtime_readonly": true, + "python_prefix": "/opt/sandboxer/runtime", + "cli_version": "2.1.266 (Claude Code)", + "private_absent": true, + "source_absent": true, + "proxy_absent": true, + "interfaces": [ + "lo" + ] + }, + "provider_requests": 2, + "cli_exit_code": 0, + "cli_is_error": false, + "cli_estimated_usd": 0.0006000000000000001, + "request_reservations": [ + { + "receipt": "5bab05cf-b2ff-427f-b85f-407116d54c11", + "run_id": "fixture-run", + "policy_sha256": "c72c3a7c69e5224c63beb60f699bbc0c28d93d4b300604cae4208e761a1b4825", + "lease_id": "d341c917fb2e1aa82e245ff5e4e6eba62d979527682d90916970cda6dd8dbb7a", + "liability_microusd": 4640000, + "state": "charged", + "observed_microusd": 500, + "created_at": "2026-09-27T15:05:05.507407+00:00" + }, + { + "receipt": "2a59879d-c6d0-496d-8f2c-0e45dddaa19f", + "run_id": "fixture-run", + "policy_sha256": "c72c3a7c69e5224c63beb60f699bbc0c28d93d4b300604cae4208e761a1b4825", + "lease_id": "d341c917fb2e1aa82e245ff5e4e6eba62d979527682d90916970cda6dd8dbb7a", + "liability_microusd": 4640000, + "state": "charged", + "observed_microusd": 500, + "created_at": "2026-09-27T15:05:05.686105+00:00" + } + ], + "route_revoked": true, + "workspace_removed": true, + "bootstrap": { + "ok": true, + "claimed": false, + "empty": true, + "run_id": "", + "ops_state": null + }, + "profile_ref": "harness.agent-dev-local@1.1.1", + "model": "claude-sonnet-5", + "request_shapes": [ + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "adaptive", + "manual_thinking_budget": false, + "sampling_parameters": {}, + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24" + ], + "tool_results": [] + }, + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "adaptive", + "manual_thinking_budget": false, + "sampling_parameters": {}, + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24" + ], + "tool_results": [ + { + "id": "fixture_git_status", + "is_error": false + } + ] + } + ], + "policy_context_tokens": 1000000, + "admission_requests": [ + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "disabled", + "top_level_fields": [ + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24", + "structured-outputs-2025-12-15" + ], + "message_roles": [ + "user" + ], + "output_config_keys": [ + "effort", + "format" + ], + "output_effort": "high", + "refusal": "beta feature not admitted" + }, + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "adaptive", + "top_level_fields": [ + "context_management", + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24" + ], + "message_roles": [ + "user" + ], + "output_config_keys": [ + "effort" + ], + "output_effort": "high" + }, + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "adaptive", + "top_level_fields": [ + "context_management", + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24" + ], + "message_roles": [ + "user", + "assistant", + "user" + ], + "output_config_keys": [ + "effort" + ], + "output_effort": "high" + } + ] + }, + "refused": { + "native_threshold_usd": 0.01, + "probe": { + "runtime_readonly": true, + "python_prefix": "/opt/sandboxer/runtime", + "cli_version": "2.1.266 (Claude Code)", + "private_absent": true, + "source_absent": true, + "proxy_absent": true, + "interfaces": [ + "lo" + ] + }, + "provider_requests": 0, + "cli_exit_code": 1, + "cli_is_error": true, + "cli_estimated_usd": 0, + "request_reservations": [], + "route_revoked": true, + "workspace_removed": true, + "bootstrap": null, + "profile_ref": "harness.agent-dev-local@1.1.1", + "model": "claude-sonnet-5", + "request_shapes": [], + "policy_context_tokens": 1000000, + "admission_requests": [ + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "disabled", + "top_level_fields": [ + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24", + "structured-outputs-2025-12-15" + ], + "message_roles": [ + "user" + ], + "output_config_keys": [ + "effort", + "format" + ], + "output_effort": "high", + "refusal": "beta feature not admitted" + }, + { + "model": "claude-sonnet-5", + "max_tokens": 64000, + "thinking_type": "adaptive", + "top_level_fields": [ + "context_management", + "max_tokens", + "messages", + "metadata", + "model", + "output_config", + "stream", + "system", + "thinking", + "tools" + ], + "betas": [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + "thinking-token-count-2026-05-13", + "context-management-2025-06-27", + "prompt-caching-scope-2026-01-05", + "mid-conversation-system-2026-04-07", + "effort-2025-11-24" + ], + "message_roles": [ + "user" + ], + "output_config_keys": [ + "effort" + ], + "output_effort": "high" + } + ] + }, + "artifact_unchanged": true, + "live_factory_attempts": 0 +} diff --git a/docs/owner-bootstrap.md b/docs/owner-bootstrap.md index 89c9f8e..586f9f6 100644 --- a/docs/owner-bootstrap.md +++ b/docs/owner-bootstrap.md @@ -71,3 +71,25 @@ positive fake stream and pre-forward insufficient-capacity refusal are separate cases. It checks unchanged artifact digest and teardown. Queue/provider/key/profile are disposable fixtures, never evidence of live admission. The project records the candidate result in `prj-helixforge-factory/evidence/2026-09-09-owner-bootstrap.json`. + + +## Current metered admission review — 2026-09-27 + +The credential catalog now configures the metered owner and companion, so the +older pending-holder paragraph above is historical. The corrected owner policy and Secrets Engine 11cc0d5 are installed after +explicit user approval; native spend/delivery approval remains open. Use +`../secrets-engine/docs/proposals/glas-metered-20260927/README.md` for exact inputs. + +The proof script requires `--profile-ref` and `--expected-model`; for the current +b6e4e8a4 artifact use `--profile-ref harness.agent-dev-local@1.1.1 +--expected-model claude-sonnet-5 --context-tokens 1000000 +--max-output-tokens 64000 --input-rate 4 --output-rate 10 +--extra-beta mid-conversation-system-2026-04-07` in addition to runtime/hash. +The fixture price input is explicit and still grants no paid authority. A passing +first-response proof does not prove a tool loop or native provider compatibility. + +Add `--tool-cycle` for a synthetic EUR 10 parent envelope and two provider +requests with an actual Bash git-status tool result. This passed on Railiance; +receipt: `docs/evidence/2026-09-27-sonnet5-tool-session-proof.json`. The fixture +uses native USD 1/max-turns 4 and total EUR 30, not the live owner spend policy. +The requested inactive EUR 10 proposal is in the factory operations directory. diff --git a/scripts/prove-metered-runtime.py b/scripts/prove-metered-runtime.py index c42aa77..d5e21c0 100644 --- a/scripts/prove-metered-runtime.py +++ b/scripts/prove-metered-runtime.py @@ -42,10 +42,10 @@ BETAS = ( DUMMY_KEY = "synthetic-provider-key-no-live-authority" -def response_bytes(): +def response_bytes(model, tool=False): events = [ {"type": "message_start", "message": {"id": "msg_fixture", "type": "message", - "role": "assistant", "model": "claude-sonnet-4-6", "content": [], + "role": "assistant", "model": model, "content": [], "stop_reason": None, "stop_sequence": None, "usage": {"input_tokens": 100, "output_tokens": 0, "cache_creation_input_tokens": 0, "cache_read_input_tokens": 0}}}, @@ -59,6 +59,10 @@ def response_bytes(): "cache_creation_input_tokens": 0, "cache_read_input_tokens": 0}}, {"type": "message_stop"}, ] + if tool: + events[1]["content_block"] = {"type": "tool_use", "id": "fixture_git_status", "name": "Bash", "input": {}} + events[2]["delta"] = {"type": "input_json_delta", "partial_json": json.dumps({"command": "git status --short", "description": "Synthetic permitted tool proof"})} + events[4]["delta"]["stop_reason"] = "tool_use" return "".join("event: " + x["type"] + "\ndata: " + json.dumps(x) + "\n\n" for x in events).encode() @@ -66,6 +70,9 @@ class FixtureServer(http.server.ThreadingHTTPServer): provider_calls = 0 queue_claims = 0 provider_key_correct = False + expected_model = "" + request_shapes = None + tool_cycle = False class Handler(http.server.BaseHTTPRequestHandler): @@ -86,19 +93,43 @@ class Handler(http.server.BaseHTTPRequestHandler): self.send_error(404) def do_POST(self): - self.rfile.read(int(self.headers.get("Content-Length", "0"))) + raw = self.rfile.read(int(self.headers.get("Content-Length", "0"))) if self.path == "/ops-runs/claim": self.server.queue_claims += 1 self.reply(b'{"items":[]}') elif self.path == "/v1/messages": self.server.provider_calls += 1 self.server.provider_key_correct = self.headers.get("x-api-key") == DUMMY_KEY - self.reply(response_bytes(), "text/event-stream") + request = json.loads(raw) + thinking = request.get("thinking", {}) + shape = {"model": request.get("model"), "max_tokens": request.get("max_tokens"), + "thinking_type": thinking.get("type"), + "manual_thinking_budget": "budget_tokens" in thinking, + "sampling_parameters": {k: request[k] for k in ("temperature", "top_p", "top_k") if k in request}, + "betas": self.headers.get("anthropic-beta", "").split(","), + "tool_results": [{"id": b.get("tool_use_id"), "is_error": b.get("is_error", False)} + for m in request.get("messages", []) + if isinstance(m.get("content"), list) + for b in m["content"] if b.get("type") == "tool_result"]} + self.server.request_shapes.append(shape) + if shape["model"] != self.server.expected_model: + self.send_error(400) + return + # Sonnet 5 rejects manual thinking and non-default sampling settings. + # Match that documented refusal rather than hiding it behind a 200. + if shape["model"] == "claude-sonnet-5" and ( + shape["thinking_type"] == "enabled" or shape["manual_thinking_budget"] + or any(value != {"temperature": 1, "top_p": 1, "top_k": 0}[key] + for key, value in shape["sampling_parameters"].items()) + ): + self.send_error(400) + return + self.reply(response_bytes(self.server.expected_model, self.server.tool_cycle and len(self.server.request_shapes) == 1), "text/event-stream") else: self.send_error(404) -def case(root, runtime, checksum, server, cap, bootstrap=False): +def case(root, runtime, checksum, server, cap, args, bootstrap=False): root.mkdir() private = root / "private"; private.mkdir(mode=0o700) source = root / "source"; source.mkdir() @@ -106,7 +137,9 @@ def case(root, runtime, checksum, server, cap, bootstrap=False): (source / "README.md").write_text("Disposable fixture. Do not edit files.\n") profiles = root / "profiles"; profiles.mkdir() catalog = ProfileCatalog() - profile, descriptor = catalog.resolve("harness.agent-dev-local@1.0.0") + profile, descriptor = catalog.resolve(args.profile_ref) + assert str(profile.ref) == args.profile_ref, "resolved profile mismatch" + assert profile.model.model == args.expected_model, "resolved model mismatch" profile = profile.model_copy(update={ "limits": ExecutionLimits(max_budget_usd=cap, max_turns=4), "operational_readiness": OperationalReadiness(status="ready", reason="disposable fixture only", @@ -121,15 +154,34 @@ def case(root, runtime, checksum, server, cap, bootstrap=False): target_repo=str(source), project="fixture-factory", profile_ref=str(profile.ref), profile_sha256=digest(profile.model_dump(mode="json")), descriptor_sha256=digest(descriptor.model_dump(mode="json")), repository_grant_id=grant.grant_id, - max_budget_usd=str(cap), max_liability_usd="5" if cap > 0.01 else "0.01", max_turns=4, - eur_per_usd="1", per_run_eur="5", daily_eur="10", total_eur="15", + max_budget_usd=str(cap), max_liability_usd=("10" if args.tool_cycle else "5") if cap > 0.01 else "0.01", max_turns=4, + eur_per_usd="1", per_run_eur="10" if args.tool_cycle else "5", daily_eur="20", total_eur="30", ) parent = SpendLedger(private / "spend.sqlite3", policy); parent.initialize() meter = RequestLedger(parent); meter.initialize() policy_path = private / "spend-policy.json" policy_path.write_text(json.dumps(policy.__dict__)); policy_path.chmod(0o600) - messages = MessagesPolicy("fixture:not-live-prices", profile.model.model, 200000, 32000, - 3, 15, allowed_betas=BETAS) + admission_requests = [] + class RecordingPolicy(MessagesPolicy): + def validate(self, data, betas): + shape = {"model": data.get("model"), "max_tokens": data.get("max_tokens"), + "thinking_type": data.get("thinking", {}).get("type"), + "top_level_fields": sorted(data), "betas": betas.split(","), + "message_roles": [m.get("role") for m in data.get("messages", [])], + "output_config_keys": sorted(data.get("output_config", {})), + "output_effort": data.get("output_config", {}).get("effort")} + admission_requests.append(shape) + try: + return super().validate(data, betas) + except Exception as exc: + shape["refusal"] = str(exc) + raise + messages = RecordingPolicy("fixture:not-live-prices", profile.model.model, + args.context_tokens, args.max_output_tokens, args.input_rate, args.output_rate, + allowed_betas=BETAS + tuple(args.extra_beta)) + server.expected_model = profile.model.model + server.request_shapes = [] + server.tool_cycle = args.tool_cycle owner_path = private / "owner.json" owner_path.write_text(json.dumps({"version": "1", "authority_ref": policy.authority_ref, "spend_policy_sha256": policy.sha256, "messages_policy": messages.__dict__, @@ -199,9 +251,12 @@ print(json.dumps({'runtime_readonly':readonly,'python_prefix':sys.prefix, assert executed.exit_code != 0 and terminal["is_error"] assert server.provider_calls == before and not meter.status() else: - assert executed.exit_code == 0 and not terminal.get("is_error") - assert server.provider_calls == before + 1 and server.provider_key_correct - assert len(meter.status()) == 1 and meter.status()[0]["state"] == "charged" + assert executed.exit_code == 0 and not terminal.get("is_error"), json.dumps({"exit_code": executed.exit_code, "subtype": terminal.get("subtype"), "request_shapes": server.request_shapes, "admission_requests": admission_requests}) + assert server.provider_calls == before + (2 if args.tool_cycle else 1) and server.provider_key_correct + assert len(meter.status()) == (2 if args.tool_cycle else 1) + assert all(row["state"] == "charged" for row in meter.status()) + if args.tool_cycle: + assert {"id": "fixture_git_status", "is_error": False} in server.request_shapes[-1]["tool_results"] token = manager._messages_route.token finally: manager.destroy(status.sandbox_id) @@ -212,7 +267,9 @@ print(json.dumps({'runtime_readonly':readonly,'python_prefix':sys.prefix, "provider_requests": server.provider_calls-before, "cli_exit_code": executed.exit_code, "cli_is_error": terminal.get("is_error"), "cli_estimated_usd": terminal.get("total_cost_usd"), "request_reservations": meter.status(), "route_revoked": True, "workspace_removed": True, - "bootstrap": bootstrap_result} + "bootstrap": bootstrap_result, "profile_ref": args.profile_ref, + "model": profile.model.model, "request_shapes": list(server.request_shapes), + "policy_context_tokens": messages.context_tokens, "admission_requests": admission_requests} assert token not in json.dumps(output) and DUMMY_KEY not in json.dumps(output) return output @@ -221,6 +278,15 @@ def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--runtime", required=True, type=Path) parser.add_argument("--sha256", required=True) + parser.add_argument("--profile-ref", required=True, + help="Exact bundled profile under test; never infer from the artifact name") + parser.add_argument("--expected-model", required=True) + parser.add_argument("--tool-cycle", action="store_true", help="Two fake-provider calls and an actual permitted git-status tool, synthetic EUR 10 envelope") + parser.add_argument("--context-tokens", type=int, default=200000) + parser.add_argument("--max-output-tokens", type=int, default=32000) + parser.add_argument("--extra-beta", action="append", default=[]) + parser.add_argument("--input-rate", type=int, default=3) + parser.add_argument("--output-rate", type=int, default=15) args = parser.parse_args() runtime = args.runtime.resolve() assert sys.flags.isolated and sys.dont_write_bytecode, "run candidate python with -I -B" @@ -238,13 +304,14 @@ def main(): server = FixtureServer(("127.0.0.1", 0), Handler) threading.Thread(target=server.serve_forever, daemon=True).start() try: - positive = case(root / "positive", runtime, args.sha256, server, 1.0, bootstrap=True) - refused = case(root / "refused", runtime, args.sha256, server, 0.01) + positive = case(root / "positive", runtime, args.sha256, server, 1.0, args, bootstrap=True) + refused = case(root / "refused", runtime, args.sha256, server, 0.01, args) finally: server.shutdown(); server.server_close() assert runtime_digest(runtime) == args.sha256, "proof mutated the candidate artifact" print(json.dumps({"scope": "standalone installed packages and real CLI/bwrap; synthetic key/provider/empty queue", "ok": True, "runtime_sha256": args.sha256, + "proof_script_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), "claude_sha256": hashlib.sha256((runtime/'bin/claude').read_bytes()).hexdigest(), "imports": imports, "packaged_definitions": True, "positive": positive, "refused": refused, "artifact_unchanged": True, "live_factory_attempts": 0}, indent=2)) diff --git a/workplans/REINAH-WP-0001-harness-foundation.md b/workplans/REINAH-WP-0001-harness-foundation.md index 73c6f6d..13bc1b7 100644 --- a/workplans/REINAH-WP-0001-harness-foundation.md +++ b/workplans/REINAH-WP-0001-harness-foundation.md @@ -2,7 +2,7 @@ id: REINAH-WP-0001 type: workplan title: "Harness Foundation: from prototype to shared runtime" -status: done +status: finished state_hub_workstream_id: "6222ad5b-d00b-515b-9fa0-de0f8ba74c28" --- diff --git a/workplans/REINAH-WP-0003-governed-runtime-integrity.md b/workplans/REINAH-WP-0003-governed-runtime-integrity.md index e98dd00..7786540 100644 --- a/workplans/REINAH-WP-0003-governed-runtime-integrity.md +++ b/workplans/REINAH-WP-0003-governed-runtime-integrity.md @@ -4,13 +4,13 @@ type: workplan title: "Governed runtime integrity and intent convergence" domain: infotech repo: rein-aharness -status: active +status: blocked flavor: implementation owner: codex topic_slug: rein-aharness priority: high created: "2026-08-23" -updated: "2026-09-09" +updated: "2026-09-27" related: - REIN-A-0004 - GLAS-IN-0002 @@ -517,7 +517,7 @@ Freedom Intelligence `d0b45acb-0002-405b-9620-cc44568170fb`, Binky ```task id: REINAH-WP-0003-T05 -status: progress +status: wait priority: high state_hub_task_id: "c4a3f08f-2874-5672-9f29-aecddd697d90" ``` @@ -691,6 +691,34 @@ runtime installation. LLM-WP-0009-T03 retains the transport integration return; T06 remains wait for admitted real-model/natural-queue evidence. No production policy, listener, credential, deployment or paid attempt was created. +### Exact profile return — 2026-09-27 + +Followed LLM-WP-0009 dependencies into the current Secrets Engine and target. +ACTIVITY-WP-0039 now supplies the dedicated live metered identity; owner config, +private ledgers and runtime b6e4e8a4 are already provisioned on Railiance. +The previous proof script always resolved profile 1.0.0, so its September 23 +receipt did not prove the new Sonnet 5 profile. Corrected the script to require +`--profile-ref` and `--expected-model`, check both, echo the selected model in +fake provider responses and record only request-shape metadata. + +The real target proof now passes with profile 1.1.1 / Sonnet 5 / CLI 2.1.266: +one fake forward, zero on insufficient capacity, bootstrap empty fake queue, +route revocation, private-state exclusion, read-only artifact and clean teardown. +Receipt: `docs/evidence/2026-09-27-sonnet5-runtime-proof.json`. +The narrowed Sonnet 5 fixture rejects manual thinking/non-default sampling. +This proves the first-response shape, not provider acceptance or a tool loop. + +Found three installed policy defects: 200k reservation against Sonnet 5's 1M +context, 32k output limit against the CLI's actual 64k request, and a missing +mid-conversation beta. Secrets Engine's corrected owner candidate and six exact +unapproved action requests live in +`../secrets-engine/docs/proposals/glas-metered-20260927/README.md`. +The USD 4.64 full hold fits only once in the existing USD 5.74 envelope; native +tool-loop success needs a separately accepted budget or another proven tighter +accounting design. T05/T06 remain open for native config/delivery, accepted +FX/tariff validity, actual queue/model/commit and recovery. No production owner +file, ledger, worker or protected runtime was changed. + ## Re-prove one governed profiled run and close residuals ```task @@ -792,3 +820,62 @@ The current catalog/CCRs, live credentials, profile readiness, host service and factory queue remain unchanged. Protected placement, provider bounds/tariffs/FX, G0 and natural execution remain. Per-run acquisition for a future continuous worker is retained here; a delivered key must not authorize an unbounded daemon lifetime. + +### Approved installation and tool-session proposal — 2026-09-27 + +The user approved the configuration/code update and separately requested preparation +of the €10 tool-session proposal. Installed Secrets Engine `11cc0d5` and corrected +owner `e0d3fb84` on Railiance; exact path/hash checks, substituted-command refusal, +standalone companion refusal and backend-free owner check pass. Standing worker, +spend limits and existing ledger are unchanged; no credentials read or paid calls. +Deployment receipt: `../secrets-engine/docs/evidence/2026-09-27-metered-owner-deployment.json`. + +Actual pinned CLI/profile 1.1.1/runtime b6e4e8a4 passes a synthetic two-request Bash +tool/result exchange and zero-forward underfunded refusal, with teardown and +unchanged artifact. Receipt: `../rein-aharness/docs/evidence/2026-09-27-sonnet5-tool-session-proof.json`. +The inactive proposal is `../prj-helixforge-factory/operations/metered-tool-session-proposal.md`: +EUR 10 run cap, USD 10 liability, EUR/USD 1.00 treatment, existing native USD 5 +threshold/daily EUR 20/total EUR 500 retained. Two USD 4.64 holds fit. Installed +0.87 FX is below the latest observed 0.876962 reference; fresh validity/FX acceptance +is required. No new grant or paid execution is authorized by proposal preparation. + +Remaining owner tasks retain their waiting status: fresh spend grant/window, +accounting continuity and replacement recipient pins; attended native per-lane +approval/delivery/revocation; then natural queue/model/tool/commit/recovery proof. +This return supersedes earlier installation-pending statements, not those gates. + +### Loose-end review — 2026-09-27 + +Reviewed every workplan in this repository and the current FI/factory owner +returns. T01–T03 are complete; T04–T06 cannot meet their remaining acceptance +criteria through changes in this repository alone. The plan is now `blocked` +and T05 joins T04/T06 in `wait`. No new tasks or workplans were created. + +Completed the remaining local CI wiring under T05: the pinned Forgejo job now +runs `make contract-test recovery-test`, so deterministic lease, crash, lock, +close-outage and replay checks run on every main push as well as at release. +Reviewed and retained the previously uncommitted exact-profile proof script, +bootstrap documentation and two synthetic receipts. The tool-session receipt's +script digest matches the committed script; the earlier single-response receipt +records its earlier script revision. Neither is a paid provider or natural-queue +success claim. + +| Task | Evidence required to resume | +| --- | --- | +| T04 | Owner-approved replacement and rollback evidence for FI and all three Binky schedules. FI-WP-0005's September 27 review still leaves T04/T05/T07/T08 waiting on grant v2, a runnable FI profile, canary and five weekday runs. Its compatibility publication is the FI-WP-0004 named grant recorded in the inventory; the September 4 local-only note above is historical. Binky has supplied no replacement evidence in this repo's inbox. Keep the dated compatibility path until those returns exist. | +| T05 | An admitted production artifact and live crash/lease/API-close/teardown recovery evidence. The synthetic pinned-artifact and local recovery proofs do not establish production recovery. Native continuous-worker credential acquisition also depends on the credential owner's per-run contract. HFACT-WP-0001-T01/T03/T04 retains fresh spend/window, accounting continuity, recipient pins and attended delivery/revocation. | +| T06 | T04/T05 completion, fresh native approvals and an actual queue/model/tool/accepted-commit run with cleanup and owner acknowledgement. HFACT-WP-0001's September 27 return still explicitly requires these gates; the inactive EUR 10 proposal does not grant paid execution. | + +All residual work remains in these existing tasks and owner records. Other +workplans already have every task done; REINAH-WP-0001's historical top-level +`done` spelling was normalized to the canonical `finished` without changing its +identity or task history. + +Validation: full suite `392 passed, 8 skipped`; isolated clean source checkouts +at every revision in `deploy/runtime-contract-lock.json`, installed using +`uv sync --frozen --no-editable --extra glas --extra llm --extra dev`, pass +the contract gate (`169 passed, 1 skipped`) and recovery gate (`131 passed, +1 skipped`). Skips are explicitly opt-in real CLI/kernel proofs, not missing +runtime packages. All sibling imports and exact source-lock checks pass. The +proof script parses and its isolated installed-package `--help` succeeds. +State Hub readback confirms the plan is `blocked` and T05 is `wait`.