Close local runtime loose ends and record external blockers
Some checks failed
Governed runtime contract / contract (push) Failing after 15s
Some checks failed
Governed runtime contract / contract (push) Failing after 15s
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e387-534d-70e3-ad53-4ea05676db8c
This commit is contained in:
parent
e18e386e0e
commit
73aaa4bcd4
8 changed files with 752 additions and 24 deletions
|
|
@ -70,4 +70,4 @@ jobs:
|
|||
"${uv_bin}" sync \
|
||||
--python "${python_bin}" \
|
||||
--frozen --no-editable --extra glas --extra llm --extra dev
|
||||
./scripts/verify-runtime-contracts.sh
|
||||
make contract-test recovery-test
|
||||
|
|
|
|||
|
|
@ -9,9 +9,9 @@
|
|||
| Kind | ID | Status | Lane | Source |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| workplan | HARNESS-WP-ADHOC-2026-09-04 | finished | — | workplans/ADHOC-2026-09-04.md |
|
||||
| workplan | REINAH-WP-0001 | done | — | workplans/REINAH-WP-0001-harness-foundation.md |
|
||||
| workplan | REINAH-WP-0001 | finished | — | workplans/REINAH-WP-0001-harness-foundation.md |
|
||||
| workplan | REINAH-WP-0002 | finished | — | workplans/REINAH-WP-0002-rename-and-glas-harness-alignment.md |
|
||||
| workplan | REINAH-WP-0003 | active | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md |
|
||||
| workplan | REINAH-WP-0003 | blocked | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md |
|
||||
| workplan | REINAH-WP-0004 | finished | — | workplans/REINAH-WP-0004-statehub-bootstrap.md |
|
||||
| workplan | REINAH-WP-0005 | finished | — | workplans/REINAH-WP-0005-ops-run-claim-loop.md |
|
||||
| workplan | REINAH-WP-0006 | finished | — | workplans/REINAH-WP-0006-binky-weekly-review-approach.md |
|
||||
|
|
@ -32,7 +32,7 @@
|
|||
| task | REINAH-WP-0003-T02 | done | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md |
|
||||
| task | REINAH-WP-0003-T03 | done | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md |
|
||||
| task | REINAH-WP-0003-T04 | wait | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md |
|
||||
| task | REINAH-WP-0003-T05 | progress | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md |
|
||||
| task | REINAH-WP-0003-T05 | wait | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md |
|
||||
| task | REINAH-WP-0003-T06 | wait | — | workplans/REINAH-WP-0003-governed-runtime-integrity.md |
|
||||
| task | REINAH-WP-0004-T01 | done | — | workplans/REINAH-WP-0004-statehub-bootstrap.md |
|
||||
| task | REINAH-WP-0004-T02 | done | — | workplans/REINAH-WP-0004-statehub-bootstrap.md |
|
||||
|
|
|
|||
242
docs/evidence/2026-09-27-sonnet5-runtime-proof.json
Normal file
242
docs/evidence/2026-09-27-sonnet5-runtime-proof.json
Normal file
|
|
@ -0,0 +1,242 @@
|
|||
{
|
||||
"scope": "standalone installed packages and real CLI/bwrap; synthetic key/provider/empty queue",
|
||||
"ok": true,
|
||||
"runtime_sha256": "b6e4e8a429393d68831c996a65c0489664205df969ac9882e581983ec2da4969",
|
||||
"proof_script_sha256": "a3cde37c6b1075caaa092248569d9b7a3e37bde467648686d70ac77ef86c89d8",
|
||||
"claude_sha256": "19842705e989393fce936804df6d2ab034860e24b8f8880357981d87ffd83fac",
|
||||
"imports": {
|
||||
"rein_aharness": "lib/python3.12/site-packages/rein_aharness/__init__.py",
|
||||
"llm_connect": "lib/python3.12/site-packages/llm_connect/__init__.py",
|
||||
"glas_harness": "lib/python3.12/site-packages/glas_harness/__init__.py",
|
||||
"sandboxer": "lib/python3.12/site-packages/sandboxer/__init__.py"
|
||||
},
|
||||
"packaged_definitions": true,
|
||||
"positive": {
|
||||
"native_threshold_usd": 1.0,
|
||||
"probe": {
|
||||
"runtime_readonly": true,
|
||||
"python_prefix": "/opt/sandboxer/runtime",
|
||||
"cli_version": "2.1.266 (Claude Code)",
|
||||
"private_absent": true,
|
||||
"source_absent": true,
|
||||
"proxy_absent": true,
|
||||
"interfaces": [
|
||||
"lo"
|
||||
]
|
||||
},
|
||||
"provider_requests": 1,
|
||||
"cli_exit_code": 0,
|
||||
"cli_is_error": false,
|
||||
"cli_estimated_usd": 0.00030000000000000003,
|
||||
"request_reservations": [
|
||||
{
|
||||
"receipt": "5519ba2f-fe1f-4928-8afb-fb765acc1bdc",
|
||||
"run_id": "fixture-run",
|
||||
"policy_sha256": "c72c3a7c69e5224c63beb60f699bbc0c28d93d4b300604cae4208e761a1b4825",
|
||||
"lease_id": "d40ebad662643b376f406fb07bf37671cb99497cc23755e3e8f97cda53b37a29",
|
||||
"liability_microusd": 4640000,
|
||||
"state": "charged",
|
||||
"observed_microusd": 500,
|
||||
"created_at": "2026-09-27T14:55:51.865416+00:00"
|
||||
}
|
||||
],
|
||||
"route_revoked": true,
|
||||
"workspace_removed": true,
|
||||
"bootstrap": {
|
||||
"ok": true,
|
||||
"claimed": false,
|
||||
"empty": true,
|
||||
"run_id": "",
|
||||
"ops_state": null
|
||||
},
|
||||
"profile_ref": "harness.agent-dev-local@1.1.1",
|
||||
"model": "claude-sonnet-5",
|
||||
"request_shapes": [
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "adaptive",
|
||||
"manual_thinking_budget": false,
|
||||
"sampling_parameters": {},
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24"
|
||||
]
|
||||
}
|
||||
],
|
||||
"policy_context_tokens": 1000000,
|
||||
"admission_requests": [
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "disabled",
|
||||
"top_level_fields": [
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24",
|
||||
"structured-outputs-2025-12-15"
|
||||
],
|
||||
"message_roles": [
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort",
|
||||
"format"
|
||||
],
|
||||
"output_effort": "high",
|
||||
"refusal": "beta feature not admitted"
|
||||
},
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "adaptive",
|
||||
"top_level_fields": [
|
||||
"context_management",
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24"
|
||||
],
|
||||
"message_roles": [
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort"
|
||||
],
|
||||
"output_effort": "high"
|
||||
}
|
||||
]
|
||||
},
|
||||
"refused": {
|
||||
"native_threshold_usd": 0.01,
|
||||
"probe": {
|
||||
"runtime_readonly": true,
|
||||
"python_prefix": "/opt/sandboxer/runtime",
|
||||
"cli_version": "2.1.266 (Claude Code)",
|
||||
"private_absent": true,
|
||||
"source_absent": true,
|
||||
"proxy_absent": true,
|
||||
"interfaces": [
|
||||
"lo"
|
||||
]
|
||||
},
|
||||
"provider_requests": 0,
|
||||
"cli_exit_code": 1,
|
||||
"cli_is_error": true,
|
||||
"cli_estimated_usd": 0,
|
||||
"request_reservations": [],
|
||||
"route_revoked": true,
|
||||
"workspace_removed": true,
|
||||
"bootstrap": null,
|
||||
"profile_ref": "harness.agent-dev-local@1.1.1",
|
||||
"model": "claude-sonnet-5",
|
||||
"request_shapes": [],
|
||||
"policy_context_tokens": 1000000,
|
||||
"admission_requests": [
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "disabled",
|
||||
"top_level_fields": [
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24",
|
||||
"structured-outputs-2025-12-15"
|
||||
],
|
||||
"message_roles": [
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort",
|
||||
"format"
|
||||
],
|
||||
"output_effort": "high",
|
||||
"refusal": "beta feature not admitted"
|
||||
},
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "adaptive",
|
||||
"top_level_fields": [
|
||||
"context_management",
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24"
|
||||
],
|
||||
"message_roles": [
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort"
|
||||
],
|
||||
"output_effort": "high"
|
||||
}
|
||||
]
|
||||
},
|
||||
"artifact_unchanged": true,
|
||||
"live_factory_attempts": 0
|
||||
}
|
||||
310
docs/evidence/2026-09-27-sonnet5-tool-session-proof.json
Normal file
310
docs/evidence/2026-09-27-sonnet5-tool-session-proof.json
Normal file
|
|
@ -0,0 +1,310 @@
|
|||
{
|
||||
"scope": "standalone installed packages and real CLI/bwrap; synthetic key/provider/empty queue",
|
||||
"ok": true,
|
||||
"runtime_sha256": "b6e4e8a429393d68831c996a65c0489664205df969ac9882e581983ec2da4969",
|
||||
"proof_script_sha256": "984c062d3e4c36671065803b1e2706f81388f31b97a8a5af9d488a31e96ca5b2",
|
||||
"claude_sha256": "19842705e989393fce936804df6d2ab034860e24b8f8880357981d87ffd83fac",
|
||||
"imports": {
|
||||
"rein_aharness": "lib/python3.12/site-packages/rein_aharness/__init__.py",
|
||||
"llm_connect": "lib/python3.12/site-packages/llm_connect/__init__.py",
|
||||
"glas_harness": "lib/python3.12/site-packages/glas_harness/__init__.py",
|
||||
"sandboxer": "lib/python3.12/site-packages/sandboxer/__init__.py"
|
||||
},
|
||||
"packaged_definitions": true,
|
||||
"positive": {
|
||||
"native_threshold_usd": 1.0,
|
||||
"probe": {
|
||||
"runtime_readonly": true,
|
||||
"python_prefix": "/opt/sandboxer/runtime",
|
||||
"cli_version": "2.1.266 (Claude Code)",
|
||||
"private_absent": true,
|
||||
"source_absent": true,
|
||||
"proxy_absent": true,
|
||||
"interfaces": [
|
||||
"lo"
|
||||
]
|
||||
},
|
||||
"provider_requests": 2,
|
||||
"cli_exit_code": 0,
|
||||
"cli_is_error": false,
|
||||
"cli_estimated_usd": 0.0006000000000000001,
|
||||
"request_reservations": [
|
||||
{
|
||||
"receipt": "5bab05cf-b2ff-427f-b85f-407116d54c11",
|
||||
"run_id": "fixture-run",
|
||||
"policy_sha256": "c72c3a7c69e5224c63beb60f699bbc0c28d93d4b300604cae4208e761a1b4825",
|
||||
"lease_id": "d341c917fb2e1aa82e245ff5e4e6eba62d979527682d90916970cda6dd8dbb7a",
|
||||
"liability_microusd": 4640000,
|
||||
"state": "charged",
|
||||
"observed_microusd": 500,
|
||||
"created_at": "2026-09-27T15:05:05.507407+00:00"
|
||||
},
|
||||
{
|
||||
"receipt": "2a59879d-c6d0-496d-8f2c-0e45dddaa19f",
|
||||
"run_id": "fixture-run",
|
||||
"policy_sha256": "c72c3a7c69e5224c63beb60f699bbc0c28d93d4b300604cae4208e761a1b4825",
|
||||
"lease_id": "d341c917fb2e1aa82e245ff5e4e6eba62d979527682d90916970cda6dd8dbb7a",
|
||||
"liability_microusd": 4640000,
|
||||
"state": "charged",
|
||||
"observed_microusd": 500,
|
||||
"created_at": "2026-09-27T15:05:05.686105+00:00"
|
||||
}
|
||||
],
|
||||
"route_revoked": true,
|
||||
"workspace_removed": true,
|
||||
"bootstrap": {
|
||||
"ok": true,
|
||||
"claimed": false,
|
||||
"empty": true,
|
||||
"run_id": "",
|
||||
"ops_state": null
|
||||
},
|
||||
"profile_ref": "harness.agent-dev-local@1.1.1",
|
||||
"model": "claude-sonnet-5",
|
||||
"request_shapes": [
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "adaptive",
|
||||
"manual_thinking_budget": false,
|
||||
"sampling_parameters": {},
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24"
|
||||
],
|
||||
"tool_results": []
|
||||
},
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "adaptive",
|
||||
"manual_thinking_budget": false,
|
||||
"sampling_parameters": {},
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24"
|
||||
],
|
||||
"tool_results": [
|
||||
{
|
||||
"id": "fixture_git_status",
|
||||
"is_error": false
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"policy_context_tokens": 1000000,
|
||||
"admission_requests": [
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "disabled",
|
||||
"top_level_fields": [
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24",
|
||||
"structured-outputs-2025-12-15"
|
||||
],
|
||||
"message_roles": [
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort",
|
||||
"format"
|
||||
],
|
||||
"output_effort": "high",
|
||||
"refusal": "beta feature not admitted"
|
||||
},
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "adaptive",
|
||||
"top_level_fields": [
|
||||
"context_management",
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24"
|
||||
],
|
||||
"message_roles": [
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort"
|
||||
],
|
||||
"output_effort": "high"
|
||||
},
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "adaptive",
|
||||
"top_level_fields": [
|
||||
"context_management",
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24"
|
||||
],
|
||||
"message_roles": [
|
||||
"user",
|
||||
"assistant",
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort"
|
||||
],
|
||||
"output_effort": "high"
|
||||
}
|
||||
]
|
||||
},
|
||||
"refused": {
|
||||
"native_threshold_usd": 0.01,
|
||||
"probe": {
|
||||
"runtime_readonly": true,
|
||||
"python_prefix": "/opt/sandboxer/runtime",
|
||||
"cli_version": "2.1.266 (Claude Code)",
|
||||
"private_absent": true,
|
||||
"source_absent": true,
|
||||
"proxy_absent": true,
|
||||
"interfaces": [
|
||||
"lo"
|
||||
]
|
||||
},
|
||||
"provider_requests": 0,
|
||||
"cli_exit_code": 1,
|
||||
"cli_is_error": true,
|
||||
"cli_estimated_usd": 0,
|
||||
"request_reservations": [],
|
||||
"route_revoked": true,
|
||||
"workspace_removed": true,
|
||||
"bootstrap": null,
|
||||
"profile_ref": "harness.agent-dev-local@1.1.1",
|
||||
"model": "claude-sonnet-5",
|
||||
"request_shapes": [],
|
||||
"policy_context_tokens": 1000000,
|
||||
"admission_requests": [
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "disabled",
|
||||
"top_level_fields": [
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24",
|
||||
"structured-outputs-2025-12-15"
|
||||
],
|
||||
"message_roles": [
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort",
|
||||
"format"
|
||||
],
|
||||
"output_effort": "high",
|
||||
"refusal": "beta feature not admitted"
|
||||
},
|
||||
{
|
||||
"model": "claude-sonnet-5",
|
||||
"max_tokens": 64000,
|
||||
"thinking_type": "adaptive",
|
||||
"top_level_fields": [
|
||||
"context_management",
|
||||
"max_tokens",
|
||||
"messages",
|
||||
"metadata",
|
||||
"model",
|
||||
"output_config",
|
||||
"stream",
|
||||
"system",
|
||||
"thinking",
|
||||
"tools"
|
||||
],
|
||||
"betas": [
|
||||
"claude-code-20250219",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
"thinking-token-count-2026-05-13",
|
||||
"context-management-2025-06-27",
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"mid-conversation-system-2026-04-07",
|
||||
"effort-2025-11-24"
|
||||
],
|
||||
"message_roles": [
|
||||
"user"
|
||||
],
|
||||
"output_config_keys": [
|
||||
"effort"
|
||||
],
|
||||
"output_effort": "high"
|
||||
}
|
||||
]
|
||||
},
|
||||
"artifact_unchanged": true,
|
||||
"live_factory_attempts": 0
|
||||
}
|
||||
|
|
@ -71,3 +71,25 @@ positive fake stream and pre-forward insufficient-capacity refusal are separate
|
|||
cases. It checks unchanged artifact digest and teardown. Queue/provider/key/profile
|
||||
are disposable fixtures, never evidence of live admission. The project records the
|
||||
candidate result in `prj-helixforge-factory/evidence/2026-09-09-owner-bootstrap.json`.
|
||||
|
||||
|
||||
## Current metered admission review — 2026-09-27
|
||||
|
||||
The credential catalog now configures the metered owner and companion, so the
|
||||
older pending-holder paragraph above is historical. The corrected owner policy and Secrets Engine 11cc0d5 are installed after
|
||||
explicit user approval; native spend/delivery approval remains open. Use
|
||||
`../secrets-engine/docs/proposals/glas-metered-20260927/README.md` for exact inputs.
|
||||
|
||||
The proof script requires `--profile-ref` and `--expected-model`; for the current
|
||||
b6e4e8a4 artifact use `--profile-ref harness.agent-dev-local@1.1.1
|
||||
--expected-model claude-sonnet-5 --context-tokens 1000000
|
||||
--max-output-tokens 64000 --input-rate 4 --output-rate 10
|
||||
--extra-beta mid-conversation-system-2026-04-07` in addition to runtime/hash.
|
||||
The fixture price input is explicit and still grants no paid authority. A passing
|
||||
first-response proof does not prove a tool loop or native provider compatibility.
|
||||
|
||||
Add `--tool-cycle` for a synthetic EUR 10 parent envelope and two provider
|
||||
requests with an actual Bash git-status tool result. This passed on Railiance;
|
||||
receipt: `docs/evidence/2026-09-27-sonnet5-tool-session-proof.json`. The fixture
|
||||
uses native USD 1/max-turns 4 and total EUR 30, not the live owner spend policy.
|
||||
The requested inactive EUR 10 proposal is in the factory operations directory.
|
||||
|
|
|
|||
|
|
@ -42,10 +42,10 @@ BETAS = (
|
|||
DUMMY_KEY = "synthetic-provider-key-no-live-authority"
|
||||
|
||||
|
||||
def response_bytes():
|
||||
def response_bytes(model, tool=False):
|
||||
events = [
|
||||
{"type": "message_start", "message": {"id": "msg_fixture", "type": "message",
|
||||
"role": "assistant", "model": "claude-sonnet-4-6", "content": [],
|
||||
"role": "assistant", "model": model, "content": [],
|
||||
"stop_reason": None, "stop_sequence": None,
|
||||
"usage": {"input_tokens": 100, "output_tokens": 0,
|
||||
"cache_creation_input_tokens": 0, "cache_read_input_tokens": 0}}},
|
||||
|
|
@ -59,6 +59,10 @@ def response_bytes():
|
|||
"cache_creation_input_tokens": 0, "cache_read_input_tokens": 0}},
|
||||
{"type": "message_stop"},
|
||||
]
|
||||
if tool:
|
||||
events[1]["content_block"] = {"type": "tool_use", "id": "fixture_git_status", "name": "Bash", "input": {}}
|
||||
events[2]["delta"] = {"type": "input_json_delta", "partial_json": json.dumps({"command": "git status --short", "description": "Synthetic permitted tool proof"})}
|
||||
events[4]["delta"]["stop_reason"] = "tool_use"
|
||||
return "".join("event: " + x["type"] + "\ndata: " + json.dumps(x) + "\n\n" for x in events).encode()
|
||||
|
||||
|
||||
|
|
@ -66,6 +70,9 @@ class FixtureServer(http.server.ThreadingHTTPServer):
|
|||
provider_calls = 0
|
||||
queue_claims = 0
|
||||
provider_key_correct = False
|
||||
expected_model = ""
|
||||
request_shapes = None
|
||||
tool_cycle = False
|
||||
|
||||
|
||||
class Handler(http.server.BaseHTTPRequestHandler):
|
||||
|
|
@ -86,19 +93,43 @@ class Handler(http.server.BaseHTTPRequestHandler):
|
|||
self.send_error(404)
|
||||
|
||||
def do_POST(self):
|
||||
self.rfile.read(int(self.headers.get("Content-Length", "0")))
|
||||
raw = self.rfile.read(int(self.headers.get("Content-Length", "0")))
|
||||
if self.path == "/ops-runs/claim":
|
||||
self.server.queue_claims += 1
|
||||
self.reply(b'{"items":[]}')
|
||||
elif self.path == "/v1/messages":
|
||||
self.server.provider_calls += 1
|
||||
self.server.provider_key_correct = self.headers.get("x-api-key") == DUMMY_KEY
|
||||
self.reply(response_bytes(), "text/event-stream")
|
||||
request = json.loads(raw)
|
||||
thinking = request.get("thinking", {})
|
||||
shape = {"model": request.get("model"), "max_tokens": request.get("max_tokens"),
|
||||
"thinking_type": thinking.get("type"),
|
||||
"manual_thinking_budget": "budget_tokens" in thinking,
|
||||
"sampling_parameters": {k: request[k] for k in ("temperature", "top_p", "top_k") if k in request},
|
||||
"betas": self.headers.get("anthropic-beta", "").split(","),
|
||||
"tool_results": [{"id": b.get("tool_use_id"), "is_error": b.get("is_error", False)}
|
||||
for m in request.get("messages", [])
|
||||
if isinstance(m.get("content"), list)
|
||||
for b in m["content"] if b.get("type") == "tool_result"]}
|
||||
self.server.request_shapes.append(shape)
|
||||
if shape["model"] != self.server.expected_model:
|
||||
self.send_error(400)
|
||||
return
|
||||
# Sonnet 5 rejects manual thinking and non-default sampling settings.
|
||||
# Match that documented refusal rather than hiding it behind a 200.
|
||||
if shape["model"] == "claude-sonnet-5" and (
|
||||
shape["thinking_type"] == "enabled" or shape["manual_thinking_budget"]
|
||||
or any(value != {"temperature": 1, "top_p": 1, "top_k": 0}[key]
|
||||
for key, value in shape["sampling_parameters"].items())
|
||||
):
|
||||
self.send_error(400)
|
||||
return
|
||||
self.reply(response_bytes(self.server.expected_model, self.server.tool_cycle and len(self.server.request_shapes) == 1), "text/event-stream")
|
||||
else:
|
||||
self.send_error(404)
|
||||
|
||||
|
||||
def case(root, runtime, checksum, server, cap, bootstrap=False):
|
||||
def case(root, runtime, checksum, server, cap, args, bootstrap=False):
|
||||
root.mkdir()
|
||||
private = root / "private"; private.mkdir(mode=0o700)
|
||||
source = root / "source"; source.mkdir()
|
||||
|
|
@ -106,7 +137,9 @@ def case(root, runtime, checksum, server, cap, bootstrap=False):
|
|||
(source / "README.md").write_text("Disposable fixture. Do not edit files.\n")
|
||||
profiles = root / "profiles"; profiles.mkdir()
|
||||
catalog = ProfileCatalog()
|
||||
profile, descriptor = catalog.resolve("harness.agent-dev-local@1.0.0")
|
||||
profile, descriptor = catalog.resolve(args.profile_ref)
|
||||
assert str(profile.ref) == args.profile_ref, "resolved profile mismatch"
|
||||
assert profile.model.model == args.expected_model, "resolved model mismatch"
|
||||
profile = profile.model_copy(update={
|
||||
"limits": ExecutionLimits(max_budget_usd=cap, max_turns=4),
|
||||
"operational_readiness": OperationalReadiness(status="ready", reason="disposable fixture only",
|
||||
|
|
@ -121,15 +154,34 @@ def case(root, runtime, checksum, server, cap, bootstrap=False):
|
|||
target_repo=str(source), project="fixture-factory", profile_ref=str(profile.ref),
|
||||
profile_sha256=digest(profile.model_dump(mode="json")),
|
||||
descriptor_sha256=digest(descriptor.model_dump(mode="json")), repository_grant_id=grant.grant_id,
|
||||
max_budget_usd=str(cap), max_liability_usd="5" if cap > 0.01 else "0.01", max_turns=4,
|
||||
eur_per_usd="1", per_run_eur="5", daily_eur="10", total_eur="15",
|
||||
max_budget_usd=str(cap), max_liability_usd=("10" if args.tool_cycle else "5") if cap > 0.01 else "0.01", max_turns=4,
|
||||
eur_per_usd="1", per_run_eur="10" if args.tool_cycle else "5", daily_eur="20", total_eur="30",
|
||||
)
|
||||
parent = SpendLedger(private / "spend.sqlite3", policy); parent.initialize()
|
||||
meter = RequestLedger(parent); meter.initialize()
|
||||
policy_path = private / "spend-policy.json"
|
||||
policy_path.write_text(json.dumps(policy.__dict__)); policy_path.chmod(0o600)
|
||||
messages = MessagesPolicy("fixture:not-live-prices", profile.model.model, 200000, 32000,
|
||||
3, 15, allowed_betas=BETAS)
|
||||
admission_requests = []
|
||||
class RecordingPolicy(MessagesPolicy):
|
||||
def validate(self, data, betas):
|
||||
shape = {"model": data.get("model"), "max_tokens": data.get("max_tokens"),
|
||||
"thinking_type": data.get("thinking", {}).get("type"),
|
||||
"top_level_fields": sorted(data), "betas": betas.split(","),
|
||||
"message_roles": [m.get("role") for m in data.get("messages", [])],
|
||||
"output_config_keys": sorted(data.get("output_config", {})),
|
||||
"output_effort": data.get("output_config", {}).get("effort")}
|
||||
admission_requests.append(shape)
|
||||
try:
|
||||
return super().validate(data, betas)
|
||||
except Exception as exc:
|
||||
shape["refusal"] = str(exc)
|
||||
raise
|
||||
messages = RecordingPolicy("fixture:not-live-prices", profile.model.model,
|
||||
args.context_tokens, args.max_output_tokens, args.input_rate, args.output_rate,
|
||||
allowed_betas=BETAS + tuple(args.extra_beta))
|
||||
server.expected_model = profile.model.model
|
||||
server.request_shapes = []
|
||||
server.tool_cycle = args.tool_cycle
|
||||
owner_path = private / "owner.json"
|
||||
owner_path.write_text(json.dumps({"version": "1", "authority_ref": policy.authority_ref,
|
||||
"spend_policy_sha256": policy.sha256, "messages_policy": messages.__dict__,
|
||||
|
|
@ -199,9 +251,12 @@ print(json.dumps({'runtime_readonly':readonly,'python_prefix':sys.prefix,
|
|||
assert executed.exit_code != 0 and terminal["is_error"]
|
||||
assert server.provider_calls == before and not meter.status()
|
||||
else:
|
||||
assert executed.exit_code == 0 and not terminal.get("is_error")
|
||||
assert server.provider_calls == before + 1 and server.provider_key_correct
|
||||
assert len(meter.status()) == 1 and meter.status()[0]["state"] == "charged"
|
||||
assert executed.exit_code == 0 and not terminal.get("is_error"), json.dumps({"exit_code": executed.exit_code, "subtype": terminal.get("subtype"), "request_shapes": server.request_shapes, "admission_requests": admission_requests})
|
||||
assert server.provider_calls == before + (2 if args.tool_cycle else 1) and server.provider_key_correct
|
||||
assert len(meter.status()) == (2 if args.tool_cycle else 1)
|
||||
assert all(row["state"] == "charged" for row in meter.status())
|
||||
if args.tool_cycle:
|
||||
assert {"id": "fixture_git_status", "is_error": False} in server.request_shapes[-1]["tool_results"]
|
||||
token = manager._messages_route.token
|
||||
finally:
|
||||
manager.destroy(status.sandbox_id)
|
||||
|
|
@ -212,7 +267,9 @@ print(json.dumps({'runtime_readonly':readonly,'python_prefix':sys.prefix,
|
|||
"provider_requests": server.provider_calls-before, "cli_exit_code": executed.exit_code,
|
||||
"cli_is_error": terminal.get("is_error"), "cli_estimated_usd": terminal.get("total_cost_usd"),
|
||||
"request_reservations": meter.status(), "route_revoked": True, "workspace_removed": True,
|
||||
"bootstrap": bootstrap_result}
|
||||
"bootstrap": bootstrap_result, "profile_ref": args.profile_ref,
|
||||
"model": profile.model.model, "request_shapes": list(server.request_shapes),
|
||||
"policy_context_tokens": messages.context_tokens, "admission_requests": admission_requests}
|
||||
assert token not in json.dumps(output) and DUMMY_KEY not in json.dumps(output)
|
||||
return output
|
||||
|
||||
|
|
@ -221,6 +278,15 @@ def main():
|
|||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--runtime", required=True, type=Path)
|
||||
parser.add_argument("--sha256", required=True)
|
||||
parser.add_argument("--profile-ref", required=True,
|
||||
help="Exact bundled profile under test; never infer from the artifact name")
|
||||
parser.add_argument("--expected-model", required=True)
|
||||
parser.add_argument("--tool-cycle", action="store_true", help="Two fake-provider calls and an actual permitted git-status tool, synthetic EUR 10 envelope")
|
||||
parser.add_argument("--context-tokens", type=int, default=200000)
|
||||
parser.add_argument("--max-output-tokens", type=int, default=32000)
|
||||
parser.add_argument("--extra-beta", action="append", default=[])
|
||||
parser.add_argument("--input-rate", type=int, default=3)
|
||||
parser.add_argument("--output-rate", type=int, default=15)
|
||||
args = parser.parse_args()
|
||||
runtime = args.runtime.resolve()
|
||||
assert sys.flags.isolated and sys.dont_write_bytecode, "run candidate python with -I -B"
|
||||
|
|
@ -238,13 +304,14 @@ def main():
|
|||
server = FixtureServer(("127.0.0.1", 0), Handler)
|
||||
threading.Thread(target=server.serve_forever, daemon=True).start()
|
||||
try:
|
||||
positive = case(root / "positive", runtime, args.sha256, server, 1.0, bootstrap=True)
|
||||
refused = case(root / "refused", runtime, args.sha256, server, 0.01)
|
||||
positive = case(root / "positive", runtime, args.sha256, server, 1.0, args, bootstrap=True)
|
||||
refused = case(root / "refused", runtime, args.sha256, server, 0.01, args)
|
||||
finally:
|
||||
server.shutdown(); server.server_close()
|
||||
assert runtime_digest(runtime) == args.sha256, "proof mutated the candidate artifact"
|
||||
print(json.dumps({"scope": "standalone installed packages and real CLI/bwrap; synthetic key/provider/empty queue",
|
||||
"ok": True, "runtime_sha256": args.sha256,
|
||||
"proof_script_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(),
|
||||
"claude_sha256": hashlib.sha256((runtime/'bin/claude').read_bytes()).hexdigest(),
|
||||
"imports": imports, "packaged_definitions": True, "positive": positive, "refused": refused,
|
||||
"artifact_unchanged": True, "live_factory_attempts": 0}, indent=2))
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
id: REINAH-WP-0001
|
||||
type: workplan
|
||||
title: "Harness Foundation: from prototype to shared runtime"
|
||||
status: done
|
||||
status: finished
|
||||
state_hub_workstream_id: "6222ad5b-d00b-515b-9fa0-de0f8ba74c28"
|
||||
---
|
||||
|
||||
|
|
|
|||
|
|
@ -4,13 +4,13 @@ type: workplan
|
|||
title: "Governed runtime integrity and intent convergence"
|
||||
domain: infotech
|
||||
repo: rein-aharness
|
||||
status: active
|
||||
status: blocked
|
||||
flavor: implementation
|
||||
owner: codex
|
||||
topic_slug: rein-aharness
|
||||
priority: high
|
||||
created: "2026-08-23"
|
||||
updated: "2026-09-09"
|
||||
updated: "2026-09-27"
|
||||
related:
|
||||
- REIN-A-0004
|
||||
- GLAS-IN-0002
|
||||
|
|
@ -517,7 +517,7 @@ Freedom Intelligence `d0b45acb-0002-405b-9620-cc44568170fb`, Binky
|
|||
|
||||
```task
|
||||
id: REINAH-WP-0003-T05
|
||||
status: progress
|
||||
status: wait
|
||||
priority: high
|
||||
state_hub_task_id: "c4a3f08f-2874-5672-9f29-aecddd697d90"
|
||||
```
|
||||
|
|
@ -691,6 +691,34 @@ runtime installation. LLM-WP-0009-T03 retains the transport integration return;
|
|||
T06 remains wait for admitted real-model/natural-queue evidence. No production
|
||||
policy, listener, credential, deployment or paid attempt was created.
|
||||
|
||||
### Exact profile return — 2026-09-27
|
||||
|
||||
Followed LLM-WP-0009 dependencies into the current Secrets Engine and target.
|
||||
ACTIVITY-WP-0039 now supplies the dedicated live metered identity; owner config,
|
||||
private ledgers and runtime b6e4e8a4 are already provisioned on Railiance.
|
||||
The previous proof script always resolved profile 1.0.0, so its September 23
|
||||
receipt did not prove the new Sonnet 5 profile. Corrected the script to require
|
||||
`--profile-ref` and `--expected-model`, check both, echo the selected model in
|
||||
fake provider responses and record only request-shape metadata.
|
||||
|
||||
The real target proof now passes with profile 1.1.1 / Sonnet 5 / CLI 2.1.266:
|
||||
one fake forward, zero on insufficient capacity, bootstrap empty fake queue,
|
||||
route revocation, private-state exclusion, read-only artifact and clean teardown.
|
||||
Receipt: `docs/evidence/2026-09-27-sonnet5-runtime-proof.json`.
|
||||
The narrowed Sonnet 5 fixture rejects manual thinking/non-default sampling.
|
||||
This proves the first-response shape, not provider acceptance or a tool loop.
|
||||
|
||||
Found three installed policy defects: 200k reservation against Sonnet 5's 1M
|
||||
context, 32k output limit against the CLI's actual 64k request, and a missing
|
||||
mid-conversation beta. Secrets Engine's corrected owner candidate and six exact
|
||||
unapproved action requests live in
|
||||
`../secrets-engine/docs/proposals/glas-metered-20260927/README.md`.
|
||||
The USD 4.64 full hold fits only once in the existing USD 5.74 envelope; native
|
||||
tool-loop success needs a separately accepted budget or another proven tighter
|
||||
accounting design. T05/T06 remain open for native config/delivery, accepted
|
||||
FX/tariff validity, actual queue/model/commit and recovery. No production owner
|
||||
file, ledger, worker or protected runtime was changed.
|
||||
|
||||
## Re-prove one governed profiled run and close residuals
|
||||
|
||||
```task
|
||||
|
|
@ -792,3 +820,62 @@ The current catalog/CCRs, live credentials, profile readiness, host service and
|
|||
factory queue remain unchanged. Protected placement, provider bounds/tariffs/FX,
|
||||
G0 and natural execution remain. Per-run acquisition for a future continuous worker
|
||||
is retained here; a delivered key must not authorize an unbounded daemon lifetime.
|
||||
|
||||
### Approved installation and tool-session proposal — 2026-09-27
|
||||
|
||||
The user approved the configuration/code update and separately requested preparation
|
||||
of the €10 tool-session proposal. Installed Secrets Engine `11cc0d5` and corrected
|
||||
owner `e0d3fb84` on Railiance; exact path/hash checks, substituted-command refusal,
|
||||
standalone companion refusal and backend-free owner check pass. Standing worker,
|
||||
spend limits and existing ledger are unchanged; no credentials read or paid calls.
|
||||
Deployment receipt: `../secrets-engine/docs/evidence/2026-09-27-metered-owner-deployment.json`.
|
||||
|
||||
Actual pinned CLI/profile 1.1.1/runtime b6e4e8a4 passes a synthetic two-request Bash
|
||||
tool/result exchange and zero-forward underfunded refusal, with teardown and
|
||||
unchanged artifact. Receipt: `../rein-aharness/docs/evidence/2026-09-27-sonnet5-tool-session-proof.json`.
|
||||
The inactive proposal is `../prj-helixforge-factory/operations/metered-tool-session-proposal.md`:
|
||||
EUR 10 run cap, USD 10 liability, EUR/USD 1.00 treatment, existing native USD 5
|
||||
threshold/daily EUR 20/total EUR 500 retained. Two USD 4.64 holds fit. Installed
|
||||
0.87 FX is below the latest observed 0.876962 reference; fresh validity/FX acceptance
|
||||
is required. No new grant or paid execution is authorized by proposal preparation.
|
||||
|
||||
Remaining owner tasks retain their waiting status: fresh spend grant/window,
|
||||
accounting continuity and replacement recipient pins; attended native per-lane
|
||||
approval/delivery/revocation; then natural queue/model/tool/commit/recovery proof.
|
||||
This return supersedes earlier installation-pending statements, not those gates.
|
||||
|
||||
### Loose-end review — 2026-09-27
|
||||
|
||||
Reviewed every workplan in this repository and the current FI/factory owner
|
||||
returns. T01–T03 are complete; T04–T06 cannot meet their remaining acceptance
|
||||
criteria through changes in this repository alone. The plan is now `blocked`
|
||||
and T05 joins T04/T06 in `wait`. No new tasks or workplans were created.
|
||||
|
||||
Completed the remaining local CI wiring under T05: the pinned Forgejo job now
|
||||
runs `make contract-test recovery-test`, so deterministic lease, crash, lock,
|
||||
close-outage and replay checks run on every main push as well as at release.
|
||||
Reviewed and retained the previously uncommitted exact-profile proof script,
|
||||
bootstrap documentation and two synthetic receipts. The tool-session receipt's
|
||||
script digest matches the committed script; the earlier single-response receipt
|
||||
records its earlier script revision. Neither is a paid provider or natural-queue
|
||||
success claim.
|
||||
|
||||
| Task | Evidence required to resume |
|
||||
| --- | --- |
|
||||
| T04 | Owner-approved replacement and rollback evidence for FI and all three Binky schedules. FI-WP-0005's September 27 review still leaves T04/T05/T07/T08 waiting on grant v2, a runnable FI profile, canary and five weekday runs. Its compatibility publication is the FI-WP-0004 named grant recorded in the inventory; the September 4 local-only note above is historical. Binky has supplied no replacement evidence in this repo's inbox. Keep the dated compatibility path until those returns exist. |
|
||||
| T05 | An admitted production artifact and live crash/lease/API-close/teardown recovery evidence. The synthetic pinned-artifact and local recovery proofs do not establish production recovery. Native continuous-worker credential acquisition also depends on the credential owner's per-run contract. HFACT-WP-0001-T01/T03/T04 retains fresh spend/window, accounting continuity, recipient pins and attended delivery/revocation. |
|
||||
| T06 | T04/T05 completion, fresh native approvals and an actual queue/model/tool/accepted-commit run with cleanup and owner acknowledgement. HFACT-WP-0001's September 27 return still explicitly requires these gates; the inactive EUR 10 proposal does not grant paid execution. |
|
||||
|
||||
All residual work remains in these existing tasks and owner records. Other
|
||||
workplans already have every task done; REINAH-WP-0001's historical top-level
|
||||
`done` spelling was normalized to the canonical `finished` without changing its
|
||||
identity or task history.
|
||||
|
||||
Validation: full suite `392 passed, 8 skipped`; isolated clean source checkouts
|
||||
at every revision in `deploy/runtime-contract-lock.json`, installed using
|
||||
`uv sync --frozen --no-editable --extra glas --extra llm --extra dev`, pass
|
||||
the contract gate (`169 passed, 1 skipped`) and recovery gate (`131 passed,
|
||||
1 skipped`). Skips are explicitly opt-in real CLI/kernel proofs, not missing
|
||||
runtime packages. All sibling imports and exact source-lock checks pass. The
|
||||
proof script parses and its isolated installed-package `--help` succeeds.
|
||||
State Hub readback confirms the plan is `blocked` and T05 is `wait`.
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue