Some checks failed
Governed runtime contract / contract (push) Failing after 27s
Assistant: codex Assistant-Model: gpt-5.6-luna Assistant-Session: 01a07ff8-19d0-7820-b4d0-1353833cb7fc
322 lines
13 KiB
Python
322 lines
13 KiB
Python
"""Agentic Claude Code adapter.
|
|
|
|
llm-connect's ClaudeCodeAdapter is a text-generation adapter (`claude
|
|
--print`, no working directory, no tool grants). An executor run needs an
|
|
*agentic* session: file edits and git commits inside the target repo, under
|
|
registered tool permission rules. This adapter subclasses it, keeping the llm-connect
|
|
LLMAdapter interface so a hosted adapter can be swapped in later, and adds:
|
|
|
|
- cwd pinned to the target repo
|
|
- acceptEdits for legacy sessions; dontAsk and an explicit tool inventory for controlled runs
|
|
- allow-list from a named tool profile (default: green-commit-only)
|
|
- optional real-time per-tool-call audit events (HARNESS-WP-0002-T03)
|
|
|
|
Claude Code executes its own tools internally in `--print` mode — it is
|
|
not possible for a caller to externally dispatch individual tool calls
|
|
(that would require abandoning Claude Code's self-contained agent model
|
|
entirely). What `--output-format stream-json --include-hook-events` does
|
|
allow: observing each tool_use/tool_result/hook event as it happens. When
|
|
`on_tool_event` is supplied, this adapter runs in that streaming mode and
|
|
invokes the callback once per event, in real time, while still returning
|
|
one aggregate `LLMResponse` at the end for interface compatibility.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
import subprocess
|
|
import threading
|
|
from pathlib import Path
|
|
from typing import Any, Callable
|
|
|
|
from llm_connect.claude_code import ClaudeCodeAdapter
|
|
from llm_connect.exceptions import LLMSubprocessError, LLMTimeoutError
|
|
from llm_connect.models import LLMResponse, RunConfig
|
|
from rein_aharness.native_limits import NativeLimitError, terminal_accounting, validate_limits
|
|
|
|
from rein_aharness.execution_cancel import (
|
|
ExecutionCancel,
|
|
ExecutionCancelled,
|
|
resolve_cancel,
|
|
)
|
|
from rein_aharness.profiles import ToolProfile, get_profile
|
|
|
|
# Backward-compatible alias for the seed profile allow-list string.
|
|
ALLOWED_TOOLS = get_profile("green-commit-only").allowed_tools
|
|
|
|
ToolEventCallback = Callable[[dict[str, Any]], None]
|
|
|
|
|
|
def _kill_process(proc: subprocess.Popen[str] | Any) -> None:
|
|
poll = getattr(proc, "poll", None)
|
|
if callable(poll):
|
|
try:
|
|
status = poll()
|
|
except Exception:
|
|
status = None
|
|
if isinstance(status, int):
|
|
return
|
|
kill = getattr(proc, "kill", None)
|
|
if callable(kill):
|
|
try:
|
|
kill()
|
|
except Exception:
|
|
return
|
|
wait = getattr(proc, "wait", None)
|
|
if callable(wait):
|
|
try:
|
|
wait(timeout=2)
|
|
except Exception:
|
|
return
|
|
|
|
|
|
def _is_tool_event(event: dict[str, Any]) -> bool:
|
|
"""True for tool_use/tool_result content blocks and hook lifecycle events.
|
|
|
|
Deliberately excludes plain assistant text messages — those aren't
|
|
tool audit events, just conversational output.
|
|
"""
|
|
event_type = event.get("type")
|
|
if event_type == "system" and str(event.get("subtype", "")).startswith("hook_"):
|
|
return True
|
|
if event_type in ("assistant", "user"):
|
|
for block in event.get("message", {}).get("content", []) or []:
|
|
if isinstance(block, dict) and block.get("type") in ("tool_use", "tool_result"):
|
|
return True
|
|
return False
|
|
|
|
|
|
class AgenticClaudeCodeAdapter(ClaudeCodeAdapter):
|
|
def __init__(
|
|
self,
|
|
workdir: Path,
|
|
*,
|
|
tool_profile: str | ToolProfile = "green-commit-only",
|
|
on_tool_event: ToolEventCallback | None = None,
|
|
cancel: ExecutionCancel | None = None,
|
|
**kwargs,
|
|
):
|
|
super().__init__(**kwargs)
|
|
self._workdir = workdir
|
|
self._on_tool_event = on_tool_event
|
|
self._cancel = cancel
|
|
self._native_config = (None, None)
|
|
self._native_cli_checked = False
|
|
if isinstance(tool_profile, ToolProfile):
|
|
self._profile = tool_profile
|
|
else:
|
|
self._profile = get_profile(tool_profile)
|
|
|
|
@property
|
|
def tool_profile(self) -> ToolProfile:
|
|
return self._profile
|
|
|
|
def _build_command(self, config: RunConfig) -> list[str]:
|
|
budget = config.model_params.get("max_budget_usd")
|
|
turns = config.model_params.get("max_turns")
|
|
validate_limits(budget, turns)
|
|
self._native_config = (budget, turns)
|
|
cmd = [
|
|
self._cli_path,
|
|
"--print",
|
|
"--permission-mode",
|
|
"acceptEdits",
|
|
"--allowedTools",
|
|
self._profile.allowed_tools,
|
|
]
|
|
if self._on_tool_event is not None:
|
|
cmd += ["--output-format", "stream-json", "--include-hook-events", "--verbose"]
|
|
elif budget is not None or turns is not None:
|
|
cmd += ["--output-format", "json"]
|
|
if budget is not None:
|
|
cmd += ["--max-budget-usd", str(budget)]
|
|
if turns is not None:
|
|
cmd += ["--max-turns", str(turns)]
|
|
if self._native_config != (None, None):
|
|
# A permission allow rule is not a tool inventory. Controlled runs
|
|
# use only the registered tools and deny every unapproved operation.
|
|
cmd[cmd.index("--permission-mode") + 1] = "dontAsk"
|
|
cmd += [
|
|
"--bare", "--setting-sources", "",
|
|
"--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}',
|
|
"--disallowedTools", "mcp__*",
|
|
"--tools", self._profile.available_tools,
|
|
"--no-session-persistence",
|
|
]
|
|
if self._model:
|
|
cmd.extend(["--model", self._model])
|
|
return cmd
|
|
|
|
def execute_prompt(self, prompt: str, config: RunConfig) -> LLMResponse:
|
|
self._preflight_budget(config)
|
|
cmd = self._build_command(config)
|
|
if self._native_config != (None, None) and not self._native_cli_checked:
|
|
check = subprocess.run([self._cli_path, "--version"], capture_output=True, text=True, timeout=10, cwd=self._workdir)
|
|
version = re.match(r"(\d+)\.(\d+)\.(\d+)(?:\s|$)", check.stdout.strip())
|
|
if check.returncode or not version or tuple(map(int, version.groups())) < (2, 1, 266):
|
|
raise NativeLimitError("native limits require verified Claude Code 2.1.266 or newer")
|
|
self._native_cli_checked = True
|
|
timeout = config.timeout_seconds or self._config.timeout_seconds
|
|
if self._on_tool_event is not None:
|
|
response = self._execute_streaming(cmd, prompt, timeout)
|
|
else:
|
|
response = self._execute_blocking(cmd, prompt, timeout)
|
|
try:
|
|
self._consume_budget(config, response)
|
|
except Exception as exc:
|
|
if self._native_config != (None, None):
|
|
raise NativeLimitError("controlled Claude run failed token accounting", cost_usd=response.metadata.get("cost_usd")) from exc
|
|
raise
|
|
return response
|
|
|
|
def _execute_blocking(self, cmd: list[str], prompt: str, timeout: int) -> LLMResponse:
|
|
proc = subprocess.Popen(
|
|
cmd,
|
|
stdin=subprocess.PIPE,
|
|
stdout=subprocess.PIPE,
|
|
stderr=subprocess.PIPE,
|
|
text=True,
|
|
cwd=self._workdir,
|
|
)
|
|
stdout, stderr = self._wait_for_process(proc, prompt, timeout)
|
|
usage, cost = {}, None
|
|
if self._native_config != (None, None):
|
|
try:
|
|
envelope = json.loads(stdout)
|
|
except (TypeError, ValueError) as exc:
|
|
raise NativeLimitError("controlled Claude run returned invalid terminal JSON") from exc
|
|
usage, cost = terminal_accounting(envelope, max_budget_usd=self._native_config[0], max_turns=self._native_config[1])
|
|
stdout = envelope.get("result", "")
|
|
if not isinstance(stdout, str):
|
|
raise NativeLimitError("controlled Claude run returned invalid result text", cost_usd=cost)
|
|
if proc.returncode != 0:
|
|
if self._native_config != (None, None):
|
|
raise NativeLimitError("controlled Claude CLI exited unsuccessfully", cost_usd=cost)
|
|
raise LLMSubprocessError(
|
|
f"claude CLI exited with code {proc.returncode}",
|
|
return_code=proc.returncode,
|
|
stderr=stderr,
|
|
)
|
|
return LLMResponse(
|
|
content=stdout,
|
|
model=self._model or "claude-code-cli",
|
|
usage=usage,
|
|
finish_reason="stop",
|
|
metadata={
|
|
"cost_usd": cost,
|
|
"provider": "claude-code-agentic",
|
|
"cli_path": self._cli_path,
|
|
"workdir": str(self._workdir),
|
|
"tool_profile": self._profile.name,
|
|
},
|
|
)
|
|
|
|
def _execute_streaming(self, cmd: list[str], prompt: str, timeout: int) -> LLMResponse:
|
|
proc = subprocess.Popen(
|
|
cmd,
|
|
stdin=subprocess.PIPE,
|
|
stdout=subprocess.PIPE,
|
|
stderr=subprocess.PIPE,
|
|
text=True,
|
|
cwd=self._workdir,
|
|
)
|
|
text_parts: list[str] = []
|
|
tool_event_count = 0
|
|
terminal_results = []
|
|
|
|
def reader() -> None:
|
|
nonlocal tool_event_count
|
|
assert proc.stdout is not None
|
|
for line in proc.stdout:
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
try:
|
|
event = json.loads(line)
|
|
except json.JSONDecodeError:
|
|
continue
|
|
if isinstance(event, dict) and event.get("type") == "result":
|
|
terminal_results.append(event)
|
|
self._handle_stream_event(event, text_parts)
|
|
if _is_tool_event(event):
|
|
tool_event_count += 1
|
|
self._on_tool_event(event)
|
|
|
|
reader_thread = threading.Thread(target=reader, daemon=True)
|
|
assert proc.stdin is not None
|
|
proc.stdin.write(prompt)
|
|
proc.stdin.close()
|
|
reader_thread.start()
|
|
returncode = self._wait_for_process(proc, None, timeout, communicate=False)
|
|
reader_thread.join(timeout=5)
|
|
stderr = proc.stderr.read() if proc.stderr else ""
|
|
|
|
usage, cost = {}, None
|
|
if self._native_config != (None, None):
|
|
if reader_thread.is_alive() or len(terminal_results) != 1:
|
|
raise NativeLimitError("controlled Claude stream requires one complete terminal result")
|
|
usage, cost = terminal_accounting(terminal_results[0], max_budget_usd=self._native_config[0], max_turns=self._native_config[1])
|
|
if returncode != 0:
|
|
if self._native_config != (None, None):
|
|
raise NativeLimitError("controlled Claude CLI exited unsuccessfully", cost_usd=cost)
|
|
raise LLMSubprocessError(
|
|
f"claude CLI exited with code {returncode}",
|
|
return_code=returncode,
|
|
stderr=stderr,
|
|
)
|
|
|
|
return LLMResponse(
|
|
content="".join(text_parts),
|
|
model=self._model or "claude-code-cli",
|
|
usage=usage,
|
|
finish_reason="stop",
|
|
metadata={
|
|
"provider": "claude-code-agentic",
|
|
"cli_path": self._cli_path,
|
|
"workdir": str(self._workdir),
|
|
"tool_profile": self._profile.name,
|
|
"tool_event_count": tool_event_count,
|
|
"cost_usd": cost,
|
|
},
|
|
)
|
|
|
|
def _wait_for_process(
|
|
self,
|
|
proc: subprocess.Popen[str],
|
|
prompt: str | None,
|
|
timeout: int,
|
|
*,
|
|
communicate: bool = True,
|
|
) -> Any:
|
|
cancel = resolve_cancel(self._cancel)
|
|
if cancel is not None:
|
|
cancel.check()
|
|
cancel.register_process(proc)
|
|
try:
|
|
if communicate:
|
|
result: Any = proc.communicate(input=prompt, timeout=timeout)
|
|
else:
|
|
result = proc.wait(timeout=timeout)
|
|
except subprocess.TimeoutExpired as exc:
|
|
_kill_process(proc)
|
|
raise LLMTimeoutError(
|
|
f"claude CLI timed out after {timeout}s", cause=exc
|
|
) from exc
|
|
except ExecutionCancelled:
|
|
raise
|
|
except Exception:
|
|
if cancel is not None and cancel.cancelled:
|
|
raise ExecutionCancelled(cancel.reason or "cancelled") from None
|
|
raise
|
|
if cancel is not None:
|
|
cancel.check()
|
|
return result
|
|
|
|
@staticmethod
|
|
def _handle_stream_event(event: dict[str, Any], text_parts: list[str]) -> None:
|
|
if event.get("type") != "assistant":
|
|
return
|
|
for block in event.get("message", {}).get("content", []) or []:
|
|
if isinstance(block, dict) and block.get("type") == "text":
|
|
text_parts.append(block["text"])
|