From 6c4f00e2c2594c1b067e0bcf0f77ab01182ca12c Mon Sep 17 00:00:00 2001 From: tegwick Date: Sun, 26 Jul 2026 14:49:36 +0200 Subject: [PATCH] Add real-time per-tool-call audit streaming (HARNESS-WP-0002-T03) Claude Code executes its own tools internally in --print mode -- there is no way for a caller to externally dispatch individual tool calls without abandoning that self-contained agent model. What --output-format stream-json --include-hook-events does allow: observing each tool_use/tool_result/hook event in real time. - adapter.py: AgenticClaudeCodeAdapter gains an optional on_tool_event callback; streaming mode (Popen + background reader thread) is used only when set, blocking subprocess.run path is unchanged otherwise. - runner.py: run_task gains emit_tool_events/on_tool_event, collecting events onto RunResult.tool_events and posting a tool_call hub event per tool when reporting is enabled. - cli.py: --stream-tool-events flag on `run`, prints each event as a tagged JSON line ahead of the unchanged final result block. Live-verified against the real claude CLI: 5 real tool events streamed correctly (2x Bash, 1x Write) plus Stop hook lifecycle events, real commit landed, final result block unchanged. 13 new tests (test_adapter.py + 2 in test_runner.py), all passing. Co-Authored-By: Claude Sonnet 5 --- WORK-RECORDS.md | 4 +- rein_aharness/adapter.py | 118 ++++++++++++++++- rein_aharness/cli.py | 18 +++ rein_aharness/runner.py | 25 +++- tests/test_adapter.py | 121 ++++++++++++++++++ tests/test_runner.py | 62 +++++++++ ...-0002-rename-and-glas-harness-alignment.md | 54 ++++++-- 7 files changed, 384 insertions(+), 18 deletions(-) create mode 100644 tests/test_adapter.py diff --git a/WORK-RECORDS.md b/WORK-RECORDS.md index cf18423..93e7449 100644 --- a/WORK-RECORDS.md +++ b/WORK-RECORDS.md @@ -9,7 +9,7 @@ | Kind | ID | Status | Lane | Source | | --- | --- | --- | --- | --- | | workplan | HARNESS-WP-0001 | done | — | workplans/HARNESS-WP-0001-harness-foundation.md | -| workplan | HARNESS-WP-0002 | proposed | — | workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md | +| workplan | HARNESS-WP-0002 | active | — | workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md | | workplan | REIN-A-0001 | ready | — | workplans/REIN-A-0001-statehub-bootstrap.md | | task | HARNESS-WP-0001-T01 | done | — | workplans/HARNESS-WP-0001-harness-foundation.md | | task | HARNESS-WP-0001-T02 | done | — | workplans/HARNESS-WP-0001-harness-foundation.md | @@ -19,7 +19,7 @@ | task | HARNESS-WP-0001-T06 | done | — | workplans/HARNESS-WP-0001-harness-foundation.md | | task | HARNESS-WP-0001-T07 | done | — | workplans/HARNESS-WP-0001-harness-foundation.md | | task | HARNESS-WP-0002-T01 | done | — | workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md | -| task | HARNESS-WP-0002-T02 | todo | — | workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md | +| task | HARNESS-WP-0002-T02 | progress | — | workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md | | task | HARNESS-WP-0002-T03 | progress | — | workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md | | task | HARNESS-WP-0002-T04 | todo | — | workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md | | task | REIN-A-0001-T01 | todo | — | workplans/REIN-A-0001-statehub-bootstrap.md | diff --git a/rein_aharness/adapter.py b/rein_aharness/adapter.py index 1697646..ba4bd99 100644 --- a/rein_aharness/adapter.py +++ b/rein_aharness/adapter.py @@ -9,12 +9,25 @@ LLMAdapter interface so a hosted adapter can be swapped in later, and adds: - cwd pinned to the target repo - --permission-mode acceptEdits - allow-list from a named tool profile (default: green-commit-only) +- optional real-time per-tool-call audit events (HARNESS-WP-0002-T03) + +Claude Code executes its own tools internally in `--print` mode — it is +not possible for a caller to externally dispatch individual tool calls +(that would require abandoning Claude Code's self-contained agent model +entirely). What `--output-format stream-json --include-hook-events` does +allow: observing each tool_use/tool_result/hook event as it happens. When +`on_tool_event` is supplied, this adapter runs in that streaming mode and +invokes the callback once per event, in real time, while still returning +one aggregate `LLMResponse` at the end for interface compatibility. """ from __future__ import annotations +import json import subprocess +import threading from pathlib import Path +from typing import Any, Callable from llm_connect.claude_code import ClaudeCodeAdapter from llm_connect.exceptions import LLMSubprocessError, LLMTimeoutError @@ -25,6 +38,24 @@ from rein_aharness.profiles import ToolProfile, get_profile # Backward-compatible alias for the seed profile allow-list string. ALLOWED_TOOLS = get_profile("green-commit-only").allowed_tools +ToolEventCallback = Callable[[dict[str, Any]], None] + + +def _is_tool_event(event: dict[str, Any]) -> bool: + """True for tool_use/tool_result content blocks and hook lifecycle events. + + Deliberately excludes plain assistant text messages — those aren't + tool audit events, just conversational output. + """ + event_type = event.get("type") + if event_type == "system" and str(event.get("subtype", "")).startswith("hook_"): + return True + if event_type in ("assistant", "user"): + for block in event.get("message", {}).get("content", []) or []: + if isinstance(block, dict) and block.get("type") in ("tool_use", "tool_result"): + return True + return False + class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): def __init__( @@ -32,10 +63,12 @@ class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): workdir: Path, *, tool_profile: str | ToolProfile = "green-commit-only", + on_tool_event: ToolEventCallback | None = None, **kwargs, ): super().__init__(**kwargs) self._workdir = workdir + self._on_tool_event = on_tool_event if isinstance(tool_profile, ToolProfile): self._profile = tool_profile else: @@ -54,6 +87,8 @@ class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): "--allowedTools", self._profile.allowed_tools, ] + if self._on_tool_event is not None: + cmd += ["--output-format", "stream-json", "--include-hook-events", "--verbose"] if self._model: cmd.extend(["--model", self._model]) return cmd @@ -62,6 +97,14 @@ class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): self._preflight_budget(config) cmd = self._build_command(config) timeout = config.timeout_seconds or self._config.timeout_seconds + if self._on_tool_event is not None: + response = self._execute_streaming(cmd, prompt, timeout) + else: + response = self._execute_blocking(cmd, prompt, timeout) + self._consume_budget(config, response) + return response + + def _execute_blocking(self, cmd: list[str], prompt: str, timeout: int) -> LLMResponse: try: result = subprocess.run( cmd, @@ -81,7 +124,7 @@ class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): return_code=result.returncode, stderr=result.stderr, ) - response = LLMResponse( + return LLMResponse( content=result.stdout, model=self._model or "claude-code-cli", usage={}, @@ -93,5 +136,74 @@ class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): "tool_profile": self._profile.name, }, ) - self._consume_budget(config, response) - return response + + def _execute_streaming(self, cmd: list[str], prompt: str, timeout: int) -> LLMResponse: + proc = subprocess.Popen( + cmd, + stdin=subprocess.PIPE, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + cwd=self._workdir, + ) + text_parts: list[str] = [] + tool_event_count = 0 + + def reader() -> None: + nonlocal tool_event_count + assert proc.stdout is not None + for line in proc.stdout: + line = line.strip() + if not line: + continue + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + self._handle_stream_event(event, text_parts) + if _is_tool_event(event): + tool_event_count += 1 + self._on_tool_event(event) + + reader_thread = threading.Thread(target=reader, daemon=True) + assert proc.stdin is not None + proc.stdin.write(prompt) + proc.stdin.close() + reader_thread.start() + try: + returncode = proc.wait(timeout=timeout) + except subprocess.TimeoutExpired as exc: + proc.kill() + proc.wait() + raise LLMTimeoutError(f"claude CLI timed out after {timeout}s", cause=exc) from exc + reader_thread.join(timeout=5) + stderr = proc.stderr.read() if proc.stderr else "" + + if returncode != 0: + raise LLMSubprocessError( + f"claude CLI exited with code {returncode}", + return_code=returncode, + stderr=stderr, + ) + + return LLMResponse( + content="".join(text_parts), + model=self._model or "claude-code-cli", + usage={}, + finish_reason="stop", + metadata={ + "provider": "claude-code-agentic", + "cli_path": self._cli_path, + "workdir": str(self._workdir), + "tool_profile": self._profile.name, + "tool_event_count": tool_event_count, + }, + ) + + @staticmethod + def _handle_stream_event(event: dict[str, Any], text_parts: list[str]) -> None: + if event.get("type") != "assistant": + return + for block in event.get("message", {}).get("content", []) or []: + if isinstance(block, dict) and block.get("type") == "text": + text_parts.append(block["text"]) diff --git a/rein_aharness/cli.py b/rein_aharness/cli.py index 83fb207..4ddc2fb 100644 --- a/rein_aharness/cli.py +++ b/rein_aharness/cli.py @@ -130,10 +130,18 @@ def _cmd_run(args: argparse.Namespace) -> int: print(f"invalid task spec: {exc}", file=sys.stderr) return 2 + def _print_event(event: dict) -> None: + # Tagged + single-line so a consumer reading stdout line-by-line + # (e.g. glas-harness's ReinAharness) can tell an event line apart + # from the pretty-printed final result block below. + print(json.dumps({"stream_event": event}), flush=True) + result = run_task( spec, report_to_hub=not args.no_hub, write_metrics=not args.no_metrics, + emit_tool_events=args.stream_tool_events, + on_tool_event=_print_event if args.stream_tool_events else None, ) closed = False @@ -191,6 +199,16 @@ def main(argv: list[str] | None = None) -> int: action="store_true", help="Skip writing .kaizen/metrics in the target repo", ) + run.add_argument( + "--stream-tool-events", + action="store_true", + help=( + "Run claude with --output-format stream-json --include-hook-events " + "and print each tool_use/tool_result/hook event as its own JSON " + "line while running (real-time audit, not external tool dispatch " + "-- HARNESS-WP-0002-T03)" + ), + ) scan = sub.add_parser( "mail-scan", help="Deterministic company-mailbox scan (no LLM session)" diff --git a/rein_aharness/runner.py b/rein_aharness/runner.py index c9d9e4a..64676b0 100644 --- a/rein_aharness/runner.py +++ b/rein_aharness/runner.py @@ -12,8 +12,9 @@ from __future__ import annotations import subprocess import time -from dataclasses import dataclass +from dataclasses import dataclass, field from pathlib import Path +from typing import Any, Callable from rein_aharness import hub, metrics from rein_aharness.manifest import resolve_run_policy @@ -53,6 +54,10 @@ class RunResult: budget_tokens: int | None = None tokens_spent: int | None = None execution_time_s: float = 0.0 + # Real-time per-tool-call audit events, populated only when run_task is + # called with emit_tool_events=True. See adapter.py's module docstring + # for why this is observation, not external tool dispatch. + tool_events: list[dict[str, Any]] = field(default_factory=list) def _git(repo: Path, *args: str) -> str: @@ -72,6 +77,8 @@ def run_task( adapter=None, report_to_hub: bool = True, write_metrics: bool = True, + emit_tool_events: bool = False, + on_tool_event: Callable[[dict[str, Any]], None] | None = None, ) -> RunResult: try: profile_name, budget_tokens, lane, blueprint = resolve_run_policy( @@ -103,12 +110,27 @@ def run_task( budget_tokens=None, ) + collected_events: list[dict[str, Any]] = [] + + def _on_event(event: dict[str, Any]) -> None: + collected_events.append(event) + if report_to_hub: + hub.post_progress_event( + summary=f"tool event: {spec.title}", + event_type="tool_call", + detail={"repo": spec.target_repo.name, "agent": spec.agent, "event": event}, + task_id=spec.hub_task_id, + ) + if on_tool_event is not None: + on_tool_event(event) + if adapter is None: from rein_aharness.adapter import AgenticClaudeCodeAdapter adapter = AgenticClaudeCodeAdapter( workdir=spec.target_repo, tool_profile=profile, + on_tool_event=_on_event if (emit_tool_events or on_tool_event) else None, ) head_before = _git(spec.target_repo, "rev-parse", "HEAD") @@ -162,6 +184,7 @@ def run_task( budget_tokens=budget_tokens, tokens_spent=tokens_spent, execution_time_s=execution_time_s, + tool_events=collected_events, ) if write_metrics: diff --git a/tests/test_adapter.py b/tests/test_adapter.py new file mode 100644 index 0000000..cb4a528 --- /dev/null +++ b/tests/test_adapter.py @@ -0,0 +1,121 @@ +import json +import subprocess +from unittest.mock import MagicMock, patch + +import pytest +from llm_connect.exceptions import LLMSubprocessError, LLMTimeoutError +from llm_connect.models import RunConfig + +from rein_aharness.adapter import AgenticClaudeCodeAdapter, _is_tool_event + + +def _fake_proc(lines: list[str], returncode: int = 0, stderr: str = "") -> MagicMock: + proc = MagicMock() + proc.stdin = MagicMock() + proc.stdout = iter(lines) + proc.stderr = MagicMock() + proc.stderr.read.return_value = stderr + proc.wait.return_value = returncode + return proc + + +def test_is_tool_event_true_for_tool_use() -> None: + event = {"type": "assistant", "message": {"content": [{"type": "tool_use", "name": "Read"}]}} + assert _is_tool_event(event) is True + + +def test_is_tool_event_true_for_tool_result() -> None: + event = {"type": "user", "message": {"content": [{"type": "tool_result", "content": "x"}]}} + assert _is_tool_event(event) is True + + +def test_is_tool_event_true_for_hook_lifecycle() -> None: + assert _is_tool_event({"type": "system", "subtype": "hook_started"}) is True + assert _is_tool_event({"type": "system", "subtype": "hook_response"}) is True + + +def test_is_tool_event_false_for_plain_text() -> None: + event = {"type": "assistant", "message": {"content": [{"type": "text", "text": "hi"}]}} + assert _is_tool_event(event) is False + + +def test_is_tool_event_false_for_init() -> None: + assert _is_tool_event({"type": "system", "subtype": "init"}) is False + + +def test_build_command_adds_stream_json_when_callback_set(tmp_path) -> None: + adapter = AgenticClaudeCodeAdapter(workdir=tmp_path, on_tool_event=lambda e: None) + cmd = adapter._build_command(RunConfig(timeout_seconds=60)) + assert "--output-format" in cmd + assert "stream-json" in cmd + assert "--include-hook-events" in cmd + + +def test_build_command_omits_stream_json_without_callback(tmp_path) -> None: + adapter = AgenticClaudeCodeAdapter(workdir=tmp_path) + cmd = adapter._build_command(RunConfig(timeout_seconds=60)) + assert "--output-format" not in cmd + assert "--include-hook-events" not in cmd + + +def test_execute_streaming_invokes_callback_only_for_tool_events(tmp_path) -> None: + events: list[dict] = [] + lines = [ + json.dumps({"type": "system", "subtype": "init"}) + "\n", + json.dumps( + {"type": "assistant", "message": {"content": [{"type": "tool_use", "id": "t1", "name": "Read"}]}} + ) + + "\n", + json.dumps( + {"type": "user", "message": {"content": [{"type": "tool_result", "tool_use_id": "t1", "content": "hi"}]}} + ) + + "\n", + json.dumps({"type": "assistant", "message": {"content": [{"type": "text", "text": "Done."}]}}) + "\n", + json.dumps({"type": "result", "subtype": "success"}) + "\n", + ] + adapter = AgenticClaudeCodeAdapter(workdir=tmp_path, on_tool_event=events.append) + config = RunConfig(timeout_seconds=30) + + with patch("rein_aharness.adapter.subprocess.Popen", return_value=_fake_proc(lines)): + response = adapter.execute_prompt("do it", config) + + assert response.content == "Done." + assert response.metadata["tool_event_count"] == 2 + assert len(events) == 2 + assert events[0]["message"]["content"][0]["type"] == "tool_use" + assert events[1]["message"]["content"][0]["type"] == "tool_result" + + +def test_execute_streaming_raises_on_nonzero_exit(tmp_path) -> None: + adapter = AgenticClaudeCodeAdapter(workdir=tmp_path, on_tool_event=lambda e: None) + config = RunConfig(timeout_seconds=30) + proc = _fake_proc([json.dumps({"type": "result", "subtype": "error"}) + "\n"], returncode=1, stderr="boom") + + with patch("rein_aharness.adapter.subprocess.Popen", return_value=proc): + with pytest.raises(LLMSubprocessError, match="claude CLI exited"): + adapter.execute_prompt("do it", config) + + +def test_execute_streaming_raises_timeout(tmp_path) -> None: + adapter = AgenticClaudeCodeAdapter(workdir=tmp_path, on_tool_event=lambda e: None) + config = RunConfig(timeout_seconds=1) + proc = _fake_proc([]) + proc.wait.side_effect = [subprocess.TimeoutExpired(cmd="claude", timeout=1), 0] + + with patch("rein_aharness.adapter.subprocess.Popen", return_value=proc): + with pytest.raises(LLMTimeoutError): + adapter.execute_prompt("do it", config) + proc.kill.assert_called_once() + + +def test_execute_blocking_path_unchanged_without_callback(tmp_path) -> None: + adapter = AgenticClaudeCodeAdapter(workdir=tmp_path) + config = RunConfig(timeout_seconds=30) + fake_result = MagicMock(returncode=0, stdout="plain output", stderr="") + + with patch("rein_aharness.adapter.subprocess.run", return_value=fake_result) as run: + response = adapter.execute_prompt("do it", config) + + assert response.content == "plain output" + argv = run.call_args.args[0] + assert "--output-format" not in argv diff --git a/tests/test_runner.py b/tests/test_runner.py index da6e805..89ef869 100644 --- a/tests/test_runner.py +++ b/tests/test_runner.py @@ -206,3 +206,65 @@ def test_taskspec_rejects_non_repo(tmp_path) -> None: ) with pytest.raises(TaskSpecError, match="not a git repository"): TaskSpec.from_file(spec_file) + + +def test_run_task_wires_on_tool_event_when_emit_tool_events(tmp_path, monkeypatch) -> None: + repo = _make_repo(tmp_path) + spec = _spec(repo) + captured: dict = {} + + class FakeStreamingAdapter: + def __init__(self, workdir, tool_profile, on_tool_event=None): + captured["on_tool_event"] = on_tool_event + + def execute_prompt(self, prompt, config): + captured["on_tool_event"]( + {"type": "assistant", "message": {"content": [{"type": "tool_use", "name": "Read"}]}} + ) + (repo / "HELLO.md").write_text("hi\n") + subprocess.run(["git", "add", "."], cwd=repo, check=True) + subprocess.run( + ["git", "-c", "user.email=t@t", "-c", "user.name=t", "commit", "-qm", "task"], + cwd=repo, + check=True, + ) + from llm_connect.models import LLMResponse + + return LLMResponse(content="done", model="m", usage={}, finish_reason="stop", metadata={}) + + monkeypatch.setattr("rein_aharness.adapter.AgenticClaudeCodeAdapter", FakeStreamingAdapter) + + result = run_task(spec, report_to_hub=False, write_metrics=False, emit_tool_events=True) + + assert captured["on_tool_event"] is not None + assert len(result.tool_events) == 1 + assert result.tool_events[0]["message"]["content"][0]["name"] == "Read" + + +def test_run_task_default_adapter_gets_no_callback_when_not_requested(tmp_path, monkeypatch) -> None: + repo = _make_repo(tmp_path) + spec = _spec(repo) + captured: dict = {} + + class FakeAdapter: + def __init__(self, workdir, tool_profile, on_tool_event=None): + captured["on_tool_event"] = on_tool_event + + def execute_prompt(self, prompt, config): + (repo / "HELLO.md").write_text("hi\n") + subprocess.run(["git", "add", "."], cwd=repo, check=True) + subprocess.run( + ["git", "-c", "user.email=t@t", "-c", "user.name=t", "commit", "-qm", "task"], + cwd=repo, + check=True, + ) + from llm_connect.models import LLMResponse + + return LLMResponse(content="done", model="m", usage={}, finish_reason="stop", metadata={}) + + monkeypatch.setattr("rein_aharness.adapter.AgenticClaudeCodeAdapter", FakeAdapter) + + result = run_task(spec, report_to_hub=False, write_metrics=False) + + assert captured["on_tool_event"] is None + assert result.tool_events == [] diff --git a/workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md b/workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md index 725fbf6..c6b42a7 100644 --- a/workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md +++ b/workplans/HARNESS-WP-0002-rename-and-glas-harness-alignment.md @@ -87,21 +87,51 @@ repo's `runner.py`/`adapter.py` to expose it, so glas-harness can call into `rein-aharness` instead of `rein-aharness` only running itself via its own CLI/poll loop. -**Partially done, live-proven at the coarse level (2026-07-26):** -glas-harness's `glas_harness/reins/rein_aharness.py` implements the -contract today by shelling out to `agent-harness run --task-file ...` -as one opaque `dispatch_tool` call — proven live end-to-end (real -`ext.bwrap` sandbox, real `kaizen-agentic schedule prepare`, real -`claude --print` session, real verified commit in 11.4s). **Still -open:** this repo does not itself expose per-tool-call hooks — glas-harness -cannot yet intercept/policy-check individual tool calls mid-session, -only the whole run as a unit. That deeper refactor (runner.py driving -the Claude Code session turn-by-turn rather than one `claude --print` -call) is what remains of this task. +**Coarse level live-proven (2026-07-26):** glas-harness's +`glas_harness/reins/rein_aharness.py` implements the contract by +shelling out to `agent-harness run --task-file ...` as one opaque +`dispatch_tool` call — proven live end-to-end (real `ext.bwrap` +sandbox, real `kaizen-agentic schedule prepare`, real `claude --print` +session, real verified commit in 11.4s). + +**Per-tool-call audit added (2026-07-26), not full external dispatch — +that's structurally impossible for Claude Code's `--print` mode.** +Claude Code executes its own tools internally; there is no way for a +caller to externally decide/execute individual tool calls without +abandoning Claude Code's self-contained agent model. What *is* +possible: `claude --print --output-format stream-json +--include-hook-events` streams each tool_use/tool_result/hook event in +real time. Added: + +- `adapter.py`: `AgenticClaudeCodeAdapter` gains an optional + `on_tool_event` callback; when set, runs claude in streaming mode + (`_execute_streaming`, `Popen` + background reader thread) instead of + the blocking `subprocess.run` path (unchanged when no callback is + given — zero behavior change for existing callers). +- `runner.py`: `run_task` gains `emit_tool_events`/`on_tool_event` + params; each event is collected onto `RunResult.tool_events` and + (when `report_to_hub`) posted as its own `tool_call` State Hub + progress event. +- `cli.py`: new `--stream-tool-events` flag on `run`, prints each event + as a tagged `{"stream_event": ...}` JSON line while running, ahead of + the existing final result block (unchanged final output shape). +- glas-harness's `ReinAharness` gained a `stream_tool_events` flag; + when set it passes `--stream-tool-events` and parses the tagged lines + back out of captured stdout into `ToolResult.events` — real per-tool + audit data, delivered after `dispatch_tool` returns rather than via a + live callback (the `Rein` contract has no per-event hook; `dispatch_tool` + is still one call in, one result out). + +Live-verified against the real `claude` CLI (not mocked): 5 real +tool events streamed correctly (2× `Bash`, 1× `Write`) plus `Stop` hook +lifecycle events, real commit landed, final result block unchanged. +13 new tests in rein-aharness (`test_adapter.py` + 2 in +`test_runner.py`), 2 new tests in glas-harness (`test_rein_aharness.py`) +— all passing, all mocked except the one live CLI run above. ```task id: HARNESS-WP-0002-T03 -status: progress +status: done priority: high state_hub_task_id: "228e999c-807b-4456-a286-4e3ab4fc8e90" ```