rein-aharness/rein_aharness/adapter.py
tegwick 4ae245a88f
Some checks failed
Governed runtime contract / contract (push) Failing after 27s
Constrain controlled CLI sessions and prove native budget overshoot
Assistant: codex
Assistant-Model: gpt-5.6-luna
Assistant-Session: 01a07ff8-19d0-7820-b4d0-1353833cb7fc
2026-09-09 20:53:25 +02:00

322 lines
13 KiB
Python

"""Agentic Claude Code adapter.
llm-connect's ClaudeCodeAdapter is a text-generation adapter (`claude
--print`, no working directory, no tool grants). An executor run needs an
*agentic* session: file edits and git commits inside the target repo, under
registered tool permission rules. This adapter subclasses it, keeping the llm-connect
LLMAdapter interface so a hosted adapter can be swapped in later, and adds:
- cwd pinned to the target repo
- acceptEdits for legacy sessions; dontAsk and an explicit tool inventory for controlled runs
- allow-list from a named tool profile (default: green-commit-only)
- optional real-time per-tool-call audit events (HARNESS-WP-0002-T03)
Claude Code executes its own tools internally in `--print` mode — it is
not possible for a caller to externally dispatch individual tool calls
(that would require abandoning Claude Code's self-contained agent model
entirely). What `--output-format stream-json --include-hook-events` does
allow: observing each tool_use/tool_result/hook event as it happens. When
`on_tool_event` is supplied, this adapter runs in that streaming mode and
invokes the callback once per event, in real time, while still returning
one aggregate `LLMResponse` at the end for interface compatibility.
"""
from __future__ import annotations
import json
import re
import subprocess
import threading
from pathlib import Path
from typing import Any, Callable
from llm_connect.claude_code import ClaudeCodeAdapter
from llm_connect.exceptions import LLMSubprocessError, LLMTimeoutError
from llm_connect.models import LLMResponse, RunConfig
from rein_aharness.native_limits import NativeLimitError, terminal_accounting, validate_limits
from rein_aharness.execution_cancel import (
ExecutionCancel,
ExecutionCancelled,
resolve_cancel,
)
from rein_aharness.profiles import ToolProfile, get_profile
# Backward-compatible alias for the seed profile allow-list string.
ALLOWED_TOOLS = get_profile("green-commit-only").allowed_tools
ToolEventCallback = Callable[[dict[str, Any]], None]
def _kill_process(proc: subprocess.Popen[str] | Any) -> None:
poll = getattr(proc, "poll", None)
if callable(poll):
try:
status = poll()
except Exception:
status = None
if isinstance(status, int):
return
kill = getattr(proc, "kill", None)
if callable(kill):
try:
kill()
except Exception:
return
wait = getattr(proc, "wait", None)
if callable(wait):
try:
wait(timeout=2)
except Exception:
return
def _is_tool_event(event: dict[str, Any]) -> bool:
"""True for tool_use/tool_result content blocks and hook lifecycle events.
Deliberately excludes plain assistant text messages — those aren't
tool audit events, just conversational output.
"""
event_type = event.get("type")
if event_type == "system" and str(event.get("subtype", "")).startswith("hook_"):
return True
if event_type in ("assistant", "user"):
for block in event.get("message", {}).get("content", []) or []:
if isinstance(block, dict) and block.get("type") in ("tool_use", "tool_result"):
return True
return False
class AgenticClaudeCodeAdapter(ClaudeCodeAdapter):
def __init__(
self,
workdir: Path,
*,
tool_profile: str | ToolProfile = "green-commit-only",
on_tool_event: ToolEventCallback | None = None,
cancel: ExecutionCancel | None = None,
**kwargs,
):
super().__init__(**kwargs)
self._workdir = workdir
self._on_tool_event = on_tool_event
self._cancel = cancel
self._native_config = (None, None)
self._native_cli_checked = False
if isinstance(tool_profile, ToolProfile):
self._profile = tool_profile
else:
self._profile = get_profile(tool_profile)
@property
def tool_profile(self) -> ToolProfile:
return self._profile
def _build_command(self, config: RunConfig) -> list[str]:
budget = config.model_params.get("max_budget_usd")
turns = config.model_params.get("max_turns")
validate_limits(budget, turns)
self._native_config = (budget, turns)
cmd = [
self._cli_path,
"--print",
"--permission-mode",
"acceptEdits",
"--allowedTools",
self._profile.allowed_tools,
]
if self._on_tool_event is not None:
cmd += ["--output-format", "stream-json", "--include-hook-events", "--verbose"]
elif budget is not None or turns is not None:
cmd += ["--output-format", "json"]
if budget is not None:
cmd += ["--max-budget-usd", str(budget)]
if turns is not None:
cmd += ["--max-turns", str(turns)]
if self._native_config != (None, None):
# A permission allow rule is not a tool inventory. Controlled runs
# use only the registered tools and deny every unapproved operation.
cmd[cmd.index("--permission-mode") + 1] = "dontAsk"
cmd += [
"--bare", "--setting-sources", "",
"--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}',
"--disallowedTools", "mcp__*",
"--tools", self._profile.available_tools,
"--no-session-persistence",
]
if self._model:
cmd.extend(["--model", self._model])
return cmd
def execute_prompt(self, prompt: str, config: RunConfig) -> LLMResponse:
self._preflight_budget(config)
cmd = self._build_command(config)
if self._native_config != (None, None) and not self._native_cli_checked:
check = subprocess.run([self._cli_path, "--version"], capture_output=True, text=True, timeout=10, cwd=self._workdir)
version = re.match(r"(\d+)\.(\d+)\.(\d+)(?:\s|$)", check.stdout.strip())
if check.returncode or not version or tuple(map(int, version.groups())) < (2, 1, 266):
raise NativeLimitError("native limits require verified Claude Code 2.1.266 or newer")
self._native_cli_checked = True
timeout = config.timeout_seconds or self._config.timeout_seconds
if self._on_tool_event is not None:
response = self._execute_streaming(cmd, prompt, timeout)
else:
response = self._execute_blocking(cmd, prompt, timeout)
try:
self._consume_budget(config, response)
except Exception as exc:
if self._native_config != (None, None):
raise NativeLimitError("controlled Claude run failed token accounting", cost_usd=response.metadata.get("cost_usd")) from exc
raise
return response
def _execute_blocking(self, cmd: list[str], prompt: str, timeout: int) -> LLMResponse:
proc = subprocess.Popen(
cmd,
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
cwd=self._workdir,
)
stdout, stderr = self._wait_for_process(proc, prompt, timeout)
usage, cost = {}, None
if self._native_config != (None, None):
try:
envelope = json.loads(stdout)
except (TypeError, ValueError) as exc:
raise NativeLimitError("controlled Claude run returned invalid terminal JSON") from exc
usage, cost = terminal_accounting(envelope, max_budget_usd=self._native_config[0], max_turns=self._native_config[1])
stdout = envelope.get("result", "")
if not isinstance(stdout, str):
raise NativeLimitError("controlled Claude run returned invalid result text", cost_usd=cost)
if proc.returncode != 0:
if self._native_config != (None, None):
raise NativeLimitError("controlled Claude CLI exited unsuccessfully", cost_usd=cost)
raise LLMSubprocessError(
f"claude CLI exited with code {proc.returncode}",
return_code=proc.returncode,
stderr=stderr,
)
return LLMResponse(
content=stdout,
model=self._model or "claude-code-cli",
usage=usage,
finish_reason="stop",
metadata={
"cost_usd": cost,
"provider": "claude-code-agentic",
"cli_path": self._cli_path,
"workdir": str(self._workdir),
"tool_profile": self._profile.name,
},
)
def _execute_streaming(self, cmd: list[str], prompt: str, timeout: int) -> LLMResponse:
proc = subprocess.Popen(
cmd,
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
cwd=self._workdir,
)
text_parts: list[str] = []
tool_event_count = 0
terminal_results = []
def reader() -> None:
nonlocal tool_event_count
assert proc.stdout is not None
for line in proc.stdout:
line = line.strip()
if not line:
continue
try:
event = json.loads(line)
except json.JSONDecodeError:
continue
if isinstance(event, dict) and event.get("type") == "result":
terminal_results.append(event)
self._handle_stream_event(event, text_parts)
if _is_tool_event(event):
tool_event_count += 1
self._on_tool_event(event)
reader_thread = threading.Thread(target=reader, daemon=True)
assert proc.stdin is not None
proc.stdin.write(prompt)
proc.stdin.close()
reader_thread.start()
returncode = self._wait_for_process(proc, None, timeout, communicate=False)
reader_thread.join(timeout=5)
stderr = proc.stderr.read() if proc.stderr else ""
usage, cost = {}, None
if self._native_config != (None, None):
if reader_thread.is_alive() or len(terminal_results) != 1:
raise NativeLimitError("controlled Claude stream requires one complete terminal result")
usage, cost = terminal_accounting(terminal_results[0], max_budget_usd=self._native_config[0], max_turns=self._native_config[1])
if returncode != 0:
if self._native_config != (None, None):
raise NativeLimitError("controlled Claude CLI exited unsuccessfully", cost_usd=cost)
raise LLMSubprocessError(
f"claude CLI exited with code {returncode}",
return_code=returncode,
stderr=stderr,
)
return LLMResponse(
content="".join(text_parts),
model=self._model or "claude-code-cli",
usage=usage,
finish_reason="stop",
metadata={
"provider": "claude-code-agentic",
"cli_path": self._cli_path,
"workdir": str(self._workdir),
"tool_profile": self._profile.name,
"tool_event_count": tool_event_count,
"cost_usd": cost,
},
)
def _wait_for_process(
self,
proc: subprocess.Popen[str],
prompt: str | None,
timeout: int,
*,
communicate: bool = True,
) -> Any:
cancel = resolve_cancel(self._cancel)
if cancel is not None:
cancel.check()
cancel.register_process(proc)
try:
if communicate:
result: Any = proc.communicate(input=prompt, timeout=timeout)
else:
result = proc.wait(timeout=timeout)
except subprocess.TimeoutExpired as exc:
_kill_process(proc)
raise LLMTimeoutError(
f"claude CLI timed out after {timeout}s", cause=exc
) from exc
except ExecutionCancelled:
raise
except Exception:
if cancel is not None and cancel.cancelled:
raise ExecutionCancelled(cancel.reason or "cancelled") from None
raise
if cancel is not None:
cancel.check()
return result
@staticmethod
def _handle_stream_event(event: dict[str, Any], text_parts: list[str]) -> None:
if event.get("type") != "assistant":
return
for block in event.get("message", {}).get("content", []) or []:
if isinstance(block, dict) and block.get("type") == "text":
text_parts.append(block["text"])