rein-aharness/rein_aharness/runner.py
tegwick 20e6f381f6 feat(runtime): enforce governed mutation boundaries
Assistant: codex
Assistant-Model: gpt-5.6-sol
Assistant-Session: 01a06ba0-10aa-7ea0-b20a-4f3fac39efe9
2026-09-04 11:25:07 +02:00

368 lines
12 KiB
Python

"""One-task run orchestration.
Flow: lock target repo → resolve tool profile / budget from instance
manifest → snapshot HEAD → persona bundle → prompt → agentic session →
validate an explicit repository grant when present → metrics + hub progress
event (+ task close). Granted runs use durable external metrics so repository
acceptance remains clean. The worker never pushes; publishing is a separate,
explicitly-granted lane.
"""
from __future__ import annotations
import subprocess
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Callable
from rein_aharness import hub, metrics
from rein_aharness.execution_cancel import ExecutionCancelled
from rein_aharness.manifest import resolve_run_policy
from rein_aharness.persona import load_persona_bundle
from rein_aharness.profiles import UnknownToolProfileError, get_profile
from rein_aharness.repository_transaction import (
DirtyRepositoryError,
GitRepositoryError,
RepositoryAcceptanceError,
RepositoryBusyError,
RepositoryTransaction,
RepositoryTransactionError,
)
from rein_aharness.taskspec import TaskSpec
PROMPT_TEMPLATE = """\
You are an unattended executor session (agent persona below, if any).
Operating rules, non-negotiable:
- Work ONLY inside the current repository working directory.
- Green/Blue lane: file edits and local git add/commit only. Never push,
never touch the network, never run destructive commands.
- Tool profile for this run: {tool_profile} (lane={lane}).
- Bounded effort: complete the single task below, commit with a clear
message, then stop. If the task cannot be completed, commit nothing and
say why in your final output.
{persona}
## Task: {title}
{description}
"""
@dataclass
class RunResult:
ok: bool
committed: bool
head_before: str
head_after: str
persona_source: str
session_output: str
reason: str = ""
model: str | None = None
tool_profile: str = ""
budget_tokens: int | None = None
tokens_spent: int | None = None
execution_time_s: float = 0.0
# Real-time per-tool-call audit events, populated only when run_task is
# called with emit_tool_events=True. See adapter.py's module docstring
# for why this is observation, not external tool dispatch.
tool_events: list[dict[str, Any]] = field(default_factory=list)
transaction: dict[str, Any] | None = None
def _git(repo: Path, *args: str) -> str:
result = subprocess.run(
["git", "-C", str(repo), *args],
capture_output=True,
text=True,
timeout=60,
)
if result.returncode != 0:
raise RuntimeError(f"git {' '.join(args)} failed: {result.stderr.strip()}")
return result.stdout.strip()
def _refused_run(
*,
reason: str,
model: str | None = None,
head_before: str = "",
transaction: dict[str, Any] | None = None,
) -> RunResult:
return RunResult(
ok=False,
committed=False,
head_before=head_before,
head_after=head_before,
persona_source="none",
session_output="",
reason=reason,
tool_profile="",
budget_tokens=None,
model=model,
transaction=transaction,
)
def run_task(
spec: TaskSpec,
adapter=None,
report_to_hub: bool = True,
write_metrics: bool = True,
emit_tool_events: bool = False,
on_tool_event: Callable[[dict[str, Any]], None] | None = None,
model: str | None = None,
tool_profile_override: str | None = None,
budget_tokens_override: int | None = None,
transaction: RepositoryTransaction | None = None,
) -> RunResult:
if spec.repository_grant is not None and not write_metrics:
return _refused_run(
reason=(
"refused: repository_grant runs require durable external metrics; "
"--no-metrics is incompatible"
),
model=model,
)
try:
profile_name, budget_tokens, lane, blueprint = resolve_run_policy(
spec.target_repo, spec.agent
)
if tool_profile_override:
profile_name = tool_profile_override
if budget_tokens_override is not None:
budget_tokens = budget_tokens_override
profile = get_profile(profile_name)
except UnknownToolProfileError as exc:
return _refused_run(reason=f"refused: {exc}", model=model)
except Exception as exc:
return _refused_run(reason=f"manifest resolution failed: {exc}", model=model)
collected_events: list[dict[str, Any]] = []
def _on_event(event: dict[str, Any]) -> None:
collected_events.append(event)
if report_to_hub:
hub.post_progress_event(
summary=f"tool event: {spec.title}",
event_type="tool_call",
detail={"repo": spec.target_repo.name, "agent": spec.agent, "event": event},
task_id=spec.hub_task_id,
)
if on_tool_event is not None:
on_tool_event(event)
if adapter is None:
from rein_aharness.adapter import AgenticClaudeCodeAdapter
adapter_kwargs = {
"workdir": spec.target_repo,
"tool_profile": profile,
"on_tool_event": _on_event if (emit_tool_events or on_tool_event) else None,
}
if model is not None:
adapter_kwargs["model"] = model
adapter = AgenticClaudeCodeAdapter(**adapter_kwargs)
def _run_locked(tx: RepositoryTransaction) -> RunResult:
return _execute_locked_task(
spec,
tx,
adapter=adapter,
profile=profile,
blueprint=blueprint,
lane=lane,
budget_tokens=budget_tokens,
collected_events=collected_events,
report_to_hub=report_to_hub,
write_metrics=write_metrics,
model=model,
)
if transaction is not None:
return _run_locked(transaction)
try:
with RepositoryTransaction(
spec.target_repo,
correlation_id=spec.hub_task_id or spec.title,
) as tx:
return _run_locked(tx)
except DirtyRepositoryError as exc:
return _refused_run(
reason=f"refused: {exc}",
model=model,
head_before=exc.baseline.head,
transaction={"baseline": exc.baseline.evidence()},
)
except RepositoryBusyError as exc:
return _refused_run(reason=f"refused: {exc}", model=model)
except (GitRepositoryError, RepositoryTransactionError) as exc:
return _refused_run(reason=f"refused: {exc}", model=model)
def _execute_locked_task(
spec: TaskSpec,
tx: RepositoryTransaction,
*,
adapter: Any,
profile: Any,
blueprint: str,
lane: str | None,
budget_tokens: int | None,
collected_events: list[dict[str, Any]],
report_to_hub: bool,
write_metrics: bool,
model: str | None,
) -> RunResult:
assert tx.baseline is not None
head_before = tx.baseline.head
persona, persona_source = load_persona_bundle(blueprint, spec.target_repo)
prompt = PROMPT_TEMPLATE.format(
persona=persona or "(no persona bundle available for this run)",
title=spec.title,
description=spec.description,
tool_profile=profile.name,
lane=lane or profile.lane,
)
from llm_connect.models import BudgetTracker, RunConfig
budget_tracker = BudgetTracker(total=budget_tokens) if budget_tokens else None
config = RunConfig(
timeout_seconds=spec.timeout_seconds,
skip_if_exists=False,
budget_tracker=budget_tracker,
)
started = time.monotonic()
try:
response = adapter.execute_prompt(prompt, config)
session_output = response.content
resolved_model = response.model
session_ok = True
reason = ""
except ExecutionCancelled as exc:
session_output = ""
session_ok = False
reason = f"execution cancelled ({exc.reason})"
resolved_model = model
except Exception as exc: # adapter / budget failures must still be reported
session_output = ""
session_ok = False
reason = f"session failed: {exc}"
resolved_model = model
execution_time_s = time.monotonic() - started
head_after = _git(spec.target_repo, "rev-parse", "HEAD")
committed = head_after != head_before
if session_ok and spec.repository_grant is not None:
try:
tx.validate_acceptance(spec.repository_grant.acceptance_policy())
except RepositoryAcceptanceError as exc:
session_ok = False
reason = str(exc)
ok = session_ok and committed
if session_ok and not committed and spec.repository_grant is None:
reason = "session completed without committing"
tokens_spent = budget_tracker.spent if budget_tracker is not None else None
transaction_evidence = tx.evidence()
if spec.repository_grant is not None:
transaction_evidence["repository_grant"] = spec.repository_grant.evidence()
metric_metadata = {
"task_title": spec.title,
"tool_profile": profile.name,
"labels": list(spec.labels),
"completion_event_type": spec.completion_event_type,
}
if spec.repository_grant is not None:
metric_metadata["repository_grant_id"] = spec.repository_grant.grant_id
if write_metrics:
try:
recorder = (
metrics.record_external_execution
if spec.repository_grant is not None
else metrics.record_execution
)
recorder(
spec.target_repo,
spec.agent,
success=ok,
execution_time_s=execution_time_s,
tokens=tokens_spent,
committed=committed,
head_after=head_after,
reason=reason or None,
metadata=metric_metadata,
session_id=tx.transaction_id,
)
if spec.repository_grant is not None:
transaction_evidence["metrics"] = {
"storage": "external",
"session_id": tx.transaction_id,
"projection_ready": True,
}
except OSError as exc:
if spec.repository_grant is not None:
ok = False
reason = (
"required external metrics persistence failed "
f"({type(exc).__name__})"
)
result = RunResult(
ok=ok,
committed=committed,
head_before=head_before,
head_after=head_after,
persona_source=persona_source,
session_output=session_output,
reason=reason,
model=resolved_model,
tool_profile=profile.name,
budget_tokens=budget_tokens,
tokens_spent=tokens_spent,
execution_time_s=execution_time_s,
tool_events=collected_events,
transaction=transaction_evidence,
)
if report_to_hub:
detail = {
"repo": spec.target_repo.name,
"task_title": spec.title,
"agent": spec.agent,
"labels": spec.labels,
"persona_source": persona_source,
"committed": committed,
"head_after": head_after,
"ok": ok,
"reason": reason,
"tool_profile": profile.name,
"budget_tokens": budget_tokens,
"tokens_spent": tokens_spent,
"execution_time_s": round(execution_time_s, 3),
"repository_transaction": transaction_evidence,
}
hub.post_progress_event(
summary=f"executor run: {spec.title} ({'ok' if ok else 'failed'})",
event_type=spec.completion_event_type,
detail=detail,
task_id=spec.hub_task_id,
)
if tokens_spent is not None or budget_tokens is not None:
hub.post_token_event(
repo=spec.target_repo.name,
tokens=tokens_spent or 0,
budget=budget_tokens,
agent=spec.agent,
ok=ok,
detail={"task_title": spec.title, "tool_profile": profile.name},
)
if ok and spec.hub_task_id:
hub.close_task(spec.hub_task_id)
return result