diff --git a/README.md b/README.md index 79a284b..0fcacea 100644 --- a/README.md +++ b/README.md @@ -11,16 +11,30 @@ OpenBao/ops-warden, reporting to the Custodian State Hub. [docs/architecture.md](docs/architecture.md) - Current work: [workplans/](workplans/) -## Usage (prototype) +## Usage ``` -agent-harness run --task-file examples/task-hello-sandbox.json [--no-hub] +# Validate a consuming repo's instance manifest (.kaizen/schedule.yml) +agent-harness validate --target ~/binky-control +agent-harness validate --target ~/binky-control --strict # require harness fields + +# List named tool profiles +agent-harness profiles + +# Run exactly one task (local JSON task-file path) +agent-harness run --task-file examples/task-hello-sandbox.json [--no-hub] [--no-metrics] + +# Deterministic mailbox scan (no LLM session) agent-harness mail-scan --target-repo ~/binky-control ``` -Runs exactly one task per invocation: persona bundle -(`kaizen-agentic schedule prepare`) → bounded agentic session -(cwd-pinned, hard tool allow-list, never pushes) → commit verification → -State Hub progress event + task close. +Each `run` resolves the agent's tool profile + budget from the target +repo's instance manifest (default `green-commit-only`), loads a persona +bundle (`kaizen-agentic schedule prepare`), runs a bounded agentic +session (cwd-pinned, hard allow-list, never pushes), verifies a local +commit, writes `.kaizen/metrics`, and posts a State Hub progress event. + +Instance manifest contract: [docs/instance-manifest.md](docs/instance-manifest.md). +Example: [examples/schedule.harness.yml](examples/schedule.harness.yml). Tests: `PYTHONPATH=".:$HOME/llm-connect" python3 -m pytest tests/ -q` diff --git a/agent_harness/__init__.py b/agent_harness/__init__.py index de51285..6c571f7 100644 --- a/agent_harness/__init__.py +++ b/agent_harness/__init__.py @@ -1,10 +1,10 @@ -"""Thin executor worker (BINKY-WP-0004-T04, DEC-2026-002). +"""Shared agent runtime (DEC-2026-002 / ADR-001). -activity-core schedules and emits tasks; this worker executes exactly one -task per invocation: load persona orientation (kaizen-agentic schedule -prepare), run a bounded agentic coding session via an llm-connect adapter, -verify the session committed to the target repo, and report to the -Custodian State Hub (progress event + optional task close). +Consuming repos declare instances in `.kaizen/schedule.yml`; this package +runs them: resolve tool profile + budget, load persona orientation +(kaizen-agentic schedule prepare), run a bounded agentic session via an +llm-connect adapter, verify the local commit, write `.kaizen/metrics`, +and report to the Custodian State Hub. """ __version__ = "0.1.0" diff --git a/agent_harness/adapter.py b/agent_harness/adapter.py index a672638..8b17a40 100644 --- a/agent_harness/adapter.py +++ b/agent_harness/adapter.py @@ -8,9 +8,7 @@ LLMAdapter interface so a hosted adapter can be swapped in later, and adds: - cwd pinned to the target repo - --permission-mode acceptEdits -- an allow-list identical in spirit to binky-control/scripts/rhythm-session.sh: - read/edit tools plus local git add/commit/status/log/diff — no push, no - network, no arbitrary shell. +- allow-list from a named tool profile (default: green-commit-only) """ from __future__ import annotations @@ -22,17 +20,30 @@ from llm_connect.claude_code import ClaudeCodeAdapter from llm_connect.exceptions import LLMSubprocessError, LLMTimeoutError from llm_connect.models import LLMResponse, RunConfig -ALLOWED_TOOLS = ( - "Read,Write,Edit,Glob,Grep," - "Bash(git add:*),Bash(git commit:*),Bash(git status)," - "Bash(git log:*),Bash(git diff:*),Bash(date:*),Bash(ls:*)" -) +from agent_harness.profiles import ToolProfile, get_profile + +# Backward-compatible alias for the seed profile allow-list string. +ALLOWED_TOOLS = get_profile("green-commit-only").allowed_tools class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): - def __init__(self, workdir: Path, **kwargs): + def __init__( + self, + workdir: Path, + *, + tool_profile: str | ToolProfile = "green-commit-only", + **kwargs, + ): super().__init__(**kwargs) self._workdir = workdir + if isinstance(tool_profile, ToolProfile): + self._profile = tool_profile + else: + self._profile = get_profile(tool_profile) + + @property + def tool_profile(self) -> ToolProfile: + return self._profile def _build_command(self, config: RunConfig) -> list[str]: cmd = [ @@ -41,7 +52,7 @@ class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): "--permission-mode", "acceptEdits", "--allowedTools", - ALLOWED_TOOLS, + self._profile.allowed_tools, ] if self._model: cmd.extend(["--model", self._model]) @@ -70,7 +81,7 @@ class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): return_code=result.returncode, stderr=result.stderr, ) - return LLMResponse( + response = LLMResponse( content=result.stdout, model=self._model or "claude-code-cli", usage={}, @@ -79,5 +90,8 @@ class AgenticClaudeCodeAdapter(ClaudeCodeAdapter): "provider": "claude-code-agentic", "cli_path": self._cli_path, "workdir": str(self._workdir), + "tool_profile": self._profile.name, }, ) + self._consume_budget(config, response) + return response diff --git a/agent_harness/cli.py b/agent_harness/cli.py index 02028b7..992740f 100644 --- a/agent_harness/cli.py +++ b/agent_harness/cli.py @@ -1,21 +1,86 @@ -"""CLI: execute exactly one task spec.""" +"""CLI: validate instance manifests and execute exactly one task.""" from __future__ import annotations import argparse import json import sys +from pathlib import Path +from agent_harness.manifest import ( + ManifestError, + load_manifest_for_repo, + manifest_path, + validate_manifest, +) +from agent_harness.profiles import list_profiles from agent_harness.runner import run_task from agent_harness.taskspec import TaskSpec, TaskSpecError +def _cmd_validate(args: argparse.Namespace) -> int: + target = Path(args.target).expanduser().resolve() + path = manifest_path(target) + if not path.is_file(): + print(f"error: no instance manifest at {path}", file=sys.stderr) + return 2 + try: + manifest = load_manifest_for_repo(target) + except ManifestError as exc: + print(f"error: {exc}", file=sys.stderr) + return 2 + + errors = validate_manifest( + manifest, require_harness_fields=bool(args.strict) + ) + if errors: + print(f"invalid: {path}", file=sys.stderr) + for err in errors: + print(f" - {err}", file=sys.stderr) + return 1 + + enabled = manifest.enabled_agents() + print(f"ok: {path}") + print(f" version: {manifest.version}") + if manifest.timezone: + print(f" timezone: {manifest.timezone}") + if manifest.harness is not None: + print(f" harness major: {manifest.harness}") + print(f" agents: {len(manifest.agents)} ({len(enabled)} enabled)") + for agent in manifest.agents: + flags = [] + if not agent.enabled: + flags.append("disabled") + if agent.tool_profile: + flags.append(f"profile={agent.tool_profile}") + if agent.lane: + flags.append(f"lane={agent.lane}") + if agent.budget is not None: + flags.append(f"budget={agent.budget}") + suffix = f" [{', '.join(flags)}]" if flags else "" + print(f" - {agent.name}: {agent.cadence}{suffix}") + return 0 + + +def _cmd_profiles(_args: argparse.Namespace) -> int: + for profile in list_profiles(): + print(f"{profile.name}\tlane={profile.lane}\t{profile.description}") + return 0 + + def main(argv: list[str] | None = None) -> int: - parser = argparse.ArgumentParser(prog="executor-worker") + parser = argparse.ArgumentParser(prog="agent-harness") sub = parser.add_subparsers(dest="command", required=True) + run = sub.add_parser("run", help="Execute one task from a JSON spec file") run.add_argument("--task-file", required=True) run.add_argument("--no-hub", action="store_true", help="Skip hub reporting") + run.add_argument( + "--no-metrics", + action="store_true", + help="Skip writing .kaizen/metrics in the target repo", + ) + scan = sub.add_parser( "mail-scan", help="Deterministic company-mailbox scan (no LLM session)" ) @@ -23,11 +88,33 @@ def main(argv: list[str] | None = None) -> int: scan.add_argument("--config", default="integrations/mailbox-binky-company.yml") scan.add_argument("--out", default="mailmeta/reports") scan.add_argument("--no-hub", action="store_true", help="Skip hub reporting") + + validate = sub.add_parser( + "validate", + help="Validate instance manifest (.kaizen/schedule.yml + harness fields)", + ) + validate.add_argument( + "--target", + default=".", + help="Consuming repo root (default: cwd)", + ) + validate.add_argument( + "--strict", + action="store_true", + help="Require lane, tool_profile, and harness pin on every enabled agent", + ) + + sub.add_parser("profiles", help="List named tool profiles") + args = parser.parse_args(argv) - if args.command == "mail-scan": - from pathlib import Path + if args.command == "validate": + return _cmd_validate(args) + if args.command == "profiles": + return _cmd_profiles(args) + + if args.command == "mail-scan": from agent_harness.mailscan import run_mail_scan result = run_mail_scan( @@ -56,7 +143,11 @@ def main(argv: list[str] | None = None) -> int: print(f"invalid task spec: {exc}", file=sys.stderr) return 2 - result = run_task(spec, report_to_hub=not args.no_hub) + result = run_task( + spec, + report_to_hub=not args.no_hub, + write_metrics=not args.no_metrics, + ) print( json.dumps( { @@ -64,6 +155,10 @@ def main(argv: list[str] | None = None) -> int: "committed": result.committed, "head_after": result.head_after, "persona_source": result.persona_source, + "tool_profile": result.tool_profile, + "budget_tokens": result.budget_tokens, + "tokens_spent": result.tokens_spent, + "execution_time_s": round(result.execution_time_s, 3), "reason": result.reason, }, indent=2, diff --git a/agent_harness/hub.py b/agent_harness/hub.py index b8563dc..edcbad2 100644 --- a/agent_harness/hub.py +++ b/agent_harness/hub.py @@ -52,3 +52,48 @@ def close_task(task_id: str) -> bool: return True except httpx.HTTPError: return False + + +def post_token_event( + repo: str, + tokens: int, + *, + budget: int | None = None, + agent: str | None = None, + ok: bool = True, + detail: dict[str, Any] | None = None, +) -> bool: + """Best-effort token cost event for the hub Token Cost dashboard. + + Schema is intentionally loose: if the hub rejects the payload we + swallow the error so a metrics-schema drift never fails a run. + """ + payload: dict[str, Any] = { + "repo": repo, + "tokens": tokens, + "source": "agent-harness", + "ok": ok, + } + if budget is not None: + payload["budget"] = budget + if agent is not None: + payload["agent"] = agent + if detail: + payload["detail"] = detail + try: + resp = httpx.post( + f"{_base_url()}/token-events/upsert", + json=payload, + timeout=_TIMEOUT, + ) + if resp.status_code >= 400: + # Fallback shape used by some hub builds + resp = httpx.post( + f"{_base_url()}/token-events/", + json=payload, + timeout=_TIMEOUT, + ) + resp.raise_for_status() + return True + except httpx.HTTPError: + return False diff --git a/agent_harness/manifest.py b/agent_harness/manifest.py new file mode 100644 index 0000000..ebc7bee --- /dev/null +++ b/agent_harness/manifest.py @@ -0,0 +1,301 @@ +"""Instance manifest — declarative agent instances in consuming repos. + +Extends ADR-005 `.kaizen/schedule.yml` with harness fields. kaizen-agentic +owns the base keys (version, timezone, agents..{cadence,cron,enabled}); +agent-harness owns the extension keys (lane, tool_profile, budget, harness, +optional blueprint). Same file — no sibling manifest unless kaizen owners +later prefer separation. + +See docs/instance-manifest.md for the full contract. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +import yaml + +from agent_harness.profiles import PROFILES, UnknownToolProfileError, get_profile + +MANIFEST_RELATIVE_PATH = Path(".kaizen") / "schedule.yml" +MANIFEST_VERSION = "1" +VALID_CADENCES = ("daily", "weekly", "monthly") +VALID_LANES = ("green", "blue") +# Package major this harness release implements (0.x → 0). +HARNESS_MAJOR = 0 + + +class ManifestError(ValueError): + """Structural failure loading or parsing a manifest.""" + + +@dataclass +class AgentInstance: + """One agent instance declared in a consuming repo.""" + + name: str + cadence: str + enabled: bool = True + cron: str | None = None + # Harness extension fields (optional in file; defaults applied at load). + blueprint: str | None = None # defaults to name + lane: str | None = None + tool_profile: str | None = None + budget: int | None = None # tokens per run; None = unlimited for this field + harness: int | None = None # pinned major; None inherits repo-level + + @property + def blueprint_name(self) -> str: + return self.blueprint or self.name + + +@dataclass +class InstanceManifest: + """Parsed instance manifest (schedule.yml + harness extensions).""" + + version: str + timezone: str | None + harness: int | None # repo-level pinned harness major + agents: list[AgentInstance] = field(default_factory=list) + source_path: Path | None = None + + def agent_for(self, name: str) -> AgentInstance | None: + for agent in self.agents: + if agent.name == name: + return agent + return None + + def enabled_agents(self) -> list[AgentInstance]: + return [a for a in self.agents if a.enabled] + + def effective_harness(self, agent: AgentInstance) -> int | None: + if agent.harness is not None: + return agent.harness + return self.harness + + +def manifest_path(project_root: Path) -> Path: + return Path(project_root) / MANIFEST_RELATIVE_PATH + + +def _parse_budget(raw: Any, agent_name: str) -> int | None: + if raw is None: + return None + if isinstance(raw, bool) or not isinstance(raw, int): + raise ManifestError( + f"agent '{agent_name}': budget must be a positive integer (tokens/run)" + ) + if raw <= 0: + raise ManifestError( + f"agent '{agent_name}': budget must be a positive integer (tokens/run)" + ) + return raw + + +def _parse_harness_major(raw: Any, context: str) -> int | None: + if raw is None: + return None + if isinstance(raw, bool) or not isinstance(raw, int): + raise ManifestError(f"{context}: harness must be a non-negative integer major") + if raw < 0: + raise ManifestError(f"{context}: harness must be a non-negative integer major") + return raw + + +def parse_manifest(data: Any, source_path: Path | None = None) -> InstanceManifest: + """Parse a raw mapping into InstanceManifest. Structural errors raise.""" + if not isinstance(data, dict): + raise ManifestError("manifest must be a YAML mapping at the top level") + + version = data.get("version") + if version is None: + raise ManifestError("missing required key: version") + version = str(version) + + timezone = data.get("timezone") + if timezone is not None and not isinstance(timezone, str): + raise ManifestError("timezone must be a string") + + repo_harness = _parse_harness_major(data.get("harness"), "top-level") + + agents_raw = data.get("agents", {}) + if not isinstance(agents_raw, dict): + raise ManifestError("agents must be a mapping of agent-name -> settings") + + agents: list[AgentInstance] = [] + for name, settings in agents_raw.items(): + if settings is None: + settings = {} + if not isinstance(settings, dict): + raise ManifestError(f"agent '{name}' settings must be a mapping") + + cron = settings.get("cron") + if cron is not None and not isinstance(cron, str): + raise ManifestError(f"agent '{name}' cron must be a string") + + blueprint = settings.get("blueprint") + if blueprint is not None and not isinstance(blueprint, str): + raise ManifestError(f"agent '{name}' blueprint must be a string") + + lane = settings.get("lane") + if lane is not None and not isinstance(lane, str): + raise ManifestError(f"agent '{name}' lane must be a string") + + tool_profile = settings.get("tool_profile") + if tool_profile is not None and not isinstance(tool_profile, str): + raise ManifestError(f"agent '{name}' tool_profile must be a string") + + agents.append( + AgentInstance( + name=str(name), + cadence=str(settings.get("cadence", "")), + enabled=bool(settings.get("enabled", True)), + cron=cron, + blueprint=blueprint, + lane=lane, + tool_profile=tool_profile, + budget=_parse_budget(settings.get("budget"), str(name)), + harness=_parse_harness_major( + settings.get("harness"), f"agent '{name}'" + ), + ) + ) + + return InstanceManifest( + version=version, + timezone=timezone, + harness=repo_harness, + agents=agents, + source_path=source_path, + ) + + +def load_manifest(path: Path | str) -> InstanceManifest: + path = Path(path) + if not path.exists(): + raise ManifestError(f"manifest not found: {path}") + try: + raw = yaml.safe_load(path.read_text(encoding="utf-8")) + except yaml.YAMLError as exc: + raise ManifestError(f"invalid YAML in {path}: {exc}") from exc + return parse_manifest(raw, source_path=path) + + +def load_manifest_for_repo(project_root: Path | str) -> InstanceManifest: + return load_manifest(manifest_path(Path(project_root))) + + +def validate_manifest( + manifest: InstanceManifest, + *, + require_harness_fields: bool = False, +) -> list[str]: + """Return human-readable validation errors (empty == valid). + + When *require_harness_fields* is False (default), ADR-005-only manifests + are valid: harness extension fields are checked only when present. When + True, every enabled agent must declare lane, tool_profile, and an + effective harness pin (repo- or agent-level). + """ + errors: list[str] = [] + + if manifest.version != MANIFEST_VERSION: + errors.append( + f"unsupported version '{manifest.version}' " + f"(expected '{MANIFEST_VERSION}')" + ) + + if not manifest.agents: + errors.append("no agents declared under 'agents:'") + + seen: set[str] = set() + for agent in manifest.agents: + if agent.name in seen: + errors.append(f"duplicate agent entry: {agent.name}") + seen.add(agent.name) + + if agent.cadence not in VALID_CADENCES: + errors.append( + f"agent '{agent.name}': invalid cadence '{agent.cadence}' " + f"(expected one of {', '.join(VALID_CADENCES)})" + ) + + if agent.lane is not None and agent.lane not in VALID_LANES: + errors.append( + f"agent '{agent.name}': invalid lane '{agent.lane}' " + f"(expected one of {', '.join(VALID_LANES)})" + ) + + if agent.tool_profile is not None: + try: + profile = get_profile(agent.tool_profile) + except UnknownToolProfileError as exc: + errors.append(f"agent '{agent.name}': {exc}") + else: + if agent.lane is not None and agent.lane != profile.lane: + errors.append( + f"agent '{agent.name}': lane '{agent.lane}' does not " + f"match tool_profile '{profile.name}' (lane={profile.lane})" + ) + + if require_harness_fields and agent.enabled: + if agent.lane is None: + errors.append( + f"agent '{agent.name}': lane is required for harness runs " + f"(green|blue)" + ) + if agent.tool_profile is None: + errors.append( + f"agent '{agent.name}': tool_profile is required for harness runs" + ) + if manifest.effective_harness(agent) is None: + errors.append( + f"agent '{agent.name}': harness major pin required " + f"(set top-level harness: or agents.{agent.name}.harness)" + ) + + effective = manifest.effective_harness(agent) + if effective is not None and effective != HARNESS_MAJOR: + # Soft pin check: warn-style error so validate fails closed for + # mismatched majors (instances must upgrade deliberately). + errors.append( + f"agent '{agent.name}': pinned harness major {effective} " + f"does not match this runtime (major {HARNESS_MAJOR})" + ) + + return errors + + +def resolve_run_policy( + project_root: Path | str, + agent_name: str, + *, + default_profile: str = "green-commit-only", +) -> tuple[str, int | None, str | None]: + """Resolve (tool_profile_name, budget_tokens, lane) for a run. + + If the repo has no manifest or no entry for *agent_name*, returns the + default profile with no budget. If the entry names an unknown profile, + raises UnknownToolProfileError (refuse to run). + """ + root = Path(project_root) + path = manifest_path(root) + if not path.exists(): + get_profile(default_profile) # validate default exists + return default_profile, None, None + + manifest = load_manifest(path) + instance = manifest.agent_for(agent_name) + if instance is None or not instance.enabled: + get_profile(default_profile) + return default_profile, None, None + + profile_name = instance.tool_profile or default_profile + get_profile(profile_name) # raises if unknown + return profile_name, instance.budget, instance.lane + + +def known_profile_names() -> list[str]: + return sorted(PROFILES) diff --git a/agent_harness/metrics.py b/agent_harness/metrics.py new file mode 100644 index 0000000..be18cf6 --- /dev/null +++ b/agent_harness/metrics.py @@ -0,0 +1,160 @@ +"""Write per-run records into the target repo's `.kaizen/metrics` tree. + +Follows kaizen-agentic ADR-004 conventions so the optimization loop can +observe harness-run agents: + + .kaizen/metrics// + executions.jsonl # append-only + summary.json # regenerated on write +""" + +from __future__ import annotations + +import json +from dataclasses import asdict, dataclass, field +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + + +@dataclass +class ExecutionRecord: + timestamp: str + agent: str + success: bool + execution_time_s: float = 0.0 + session_id: str | None = None + quality_score: float | None = None + primary_metric: dict[str, Any] | None = None + metadata: dict[str, Any] = field(default_factory=dict) + # Helix / harness correlation (ADR-004 optional fields) + repo: str | None = None + tokens: int | None = None + committed: bool | None = None + head_after: str | None = None + reason: str | None = None + harness: str = "agent-harness" + + def to_json_line(self) -> str: + data = asdict(self) + # Drop Nones for a compact record; required fields always present. + compact = {k: v for k, v in data.items() if v is not None} + return json.dumps(compact, sort_keys=True) + + +def metrics_dir(project_root: Path, agent: str) -> Path: + return Path(project_root) / ".kaizen" / "metrics" / agent + + +def _utc_now() -> str: + return datetime.now(timezone.utc).replace(microsecond=0).isoformat().replace( + "+00:00", "Z" + ) + + +def _load_executions(path: Path) -> list[dict[str, Any]]: + if not path.is_file(): + return [] + records: list[dict[str, Any]] = [] + for line in path.read_text(encoding="utf-8").splitlines(): + line = line.strip() + if not line: + continue + try: + records.append(json.loads(line)) + except json.JSONDecodeError: + continue + return records + + +def _trend(values: list[float]) -> str: + if len(values) < 4: + return "stable" + mid = len(values) // 2 + early = sum(values[:mid]) / max(1, mid) + late = sum(values[mid:]) / max(1, len(values) - mid) + if late - early > 0.05: + return "up" + if early - late > 0.05: + return "down" + return "stable" + + +def regenerate_summary(agent: str, executions: list[dict[str, Any]]) -> dict[str, Any]: + count = len(executions) + successes = [e for e in executions if e.get("success")] + success_rate = (len(successes) / count) if count else 0.0 + times = [float(e["execution_time_s"]) for e in executions if "execution_time_s" in e] + qualities = [ + float(e["quality_score"]) + for e in executions + if isinstance(e.get("quality_score"), (int, float)) + ] + last_ts = executions[-1].get("timestamp") if executions else None + return { + "agent": agent, + "execution_count": count, + "success_rate": round(success_rate, 3), + "avg_quality_score": ( + round(sum(qualities) / len(qualities), 3) if qualities else None + ), + "avg_execution_time_s": ( + round(sum(times) / len(times), 3) if times else None + ), + "last_execution": last_ts, + "trend": { + "success_rate": _trend( + [1.0 if e.get("success") else 0.0 for e in executions] + ), + "quality_score": _trend(qualities) if qualities else "stable", + }, + } + + +def record_execution( + project_root: Path | str, + agent: str, + *, + success: bool, + execution_time_s: float = 0.0, + tokens: int | None = None, + committed: bool | None = None, + head_after: str | None = None, + reason: str | None = None, + metadata: dict[str, Any] | None = None, + session_id: str | None = None, +) -> Path: + """Append one execution record and regenerate summary.json. + + Returns the path to executions.jsonl. Never raises for missing parent + dirs (creates them). Callers that must not write should skip this. + """ + root = Path(project_root) + directory = metrics_dir(root, agent) + directory.mkdir(parents=True, exist_ok=True) + + record = ExecutionRecord( + timestamp=_utc_now(), + agent=agent, + success=success, + execution_time_s=float(execution_time_s), + session_id=session_id, + metadata=metadata or {}, + repo=root.name, + tokens=tokens, + committed=committed, + head_after=head_after, + reason=reason, + ) + + executions_path = directory / "executions.jsonl" + with executions_path.open("a", encoding="utf-8") as fh: + fh.write(record.to_json_line() + "\n") + + all_records = _load_executions(executions_path) + summary = regenerate_summary(agent, all_records) + (directory / "summary.json").write_text( + json.dumps(summary, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + return executions_path diff --git a/agent_harness/profiles.py b/agent_harness/profiles.py new file mode 100644 index 0000000..5867003 --- /dev/null +++ b/agent_harness/profiles.py @@ -0,0 +1,78 @@ +"""Named tool-profile registry. + +Instances declare a profile by name in the instance manifest; the harness +resolves and enforces the allow-list. Instances never enumerate tools. +Unknown profile names refuse to run. +""" + +from __future__ import annotations + +from dataclasses import dataclass + + +class UnknownToolProfileError(ValueError): + """Raised when a manifest references a profile that is not registered.""" + + +# Claude Code --allowedTools strings. No push, no network, no arbitrary shell. +_GREEN_COMMIT_TOOLS = ( + "Read,Write,Edit,Glob,Grep," + "Bash(git add:*),Bash(git commit:*),Bash(git status)," + "Bash(git log:*),Bash(git diff:*),Bash(date:*),Bash(ls:*)" +) + +# Blue-lane mail triage session: same session tools as green-commit-only. +# Credentialed IMAP scan is a deterministic pre-step outside the session +# (see mailscan.py); the session only reads reports and updates queues. +_BLUE_MAIL_TRIAGE_TOOLS = _GREEN_COMMIT_TOOLS + + +@dataclass(frozen=True) +class ToolProfile: + """A named hard allow-list for agentic sessions.""" + + name: str + description: str + allowed_tools: str + lane: str # green | blue — advisory; enforcement is the allow-list + + +PROFILES: dict[str, ToolProfile] = { + "green-commit-only": ToolProfile( + name="green-commit-only", + description=( + "Green-lane local commit: read/edit tools plus git add/commit/" + "status/log/diff. No push, no network, no arbitrary shell." + ), + allowed_tools=_GREEN_COMMIT_TOOLS, + lane="green", + ), + "blue-mail-triage": ToolProfile( + name="blue-mail-triage", + description=( + "Blue-lane mail triage session after a credentialed deterministic " + "scan: same local commit tools as green-commit-only. Credentials " + "and network stay outside the agentic session." + ), + allowed_tools=_BLUE_MAIL_TRIAGE_TOOLS, + lane="blue", + ), +} + + +def get_profile(name: str) -> ToolProfile: + """Resolve a profile by name. Raises UnknownToolProfileError if missing.""" + key = (name or "").strip() + if not key: + raise UnknownToolProfileError("tool_profile name is empty") + profile = PROFILES.get(key) + if profile is None: + known = ", ".join(sorted(PROFILES)) + raise UnknownToolProfileError( + f"unknown tool_profile '{key}' (known: {known})" + ) + return profile + + +def list_profiles() -> list[ToolProfile]: + return [PROFILES[k] for k in sorted(PROFILES)] diff --git a/agent_harness/runner.py b/agent_harness/runner.py index b54e89f..19766a2 100644 --- a/agent_harness/runner.py +++ b/agent_harness/runner.py @@ -1,21 +1,24 @@ """One-task run orchestration. -Flow: lock target repo → snapshot HEAD → persona bundle → prompt → agentic -session → verify a new commit exists → hub progress event (+ task close). -The run *fails* if the session pushed anywhere or left the repo dirty in a -way it should not — the worker never pushes; publishing is a separate, -explicitly-granted lane (see integrations/executor-worker-secrets.md in -binky-control). +Flow: lock target repo → resolve tool profile / budget from instance +manifest → snapshot HEAD → persona bundle → prompt → agentic session → +verify a new commit exists → kaizen metrics + hub progress event +(+ task close). The run *fails* if the session pushed anywhere or left +the repo dirty in a way it should not — the worker never pushes; +publishing is a separate, explicitly-granted lane. """ from __future__ import annotations import subprocess +import time from dataclasses import dataclass from pathlib import Path -from agent_harness import hub +from agent_harness import hub, metrics +from agent_harness.manifest import resolve_run_policy from agent_harness.persona import load_persona_bundle +from agent_harness.profiles import UnknownToolProfileError, get_profile from agent_harness.taskspec import TaskSpec PROMPT_TEMPLATE = """\ @@ -24,6 +27,7 @@ Operating rules, non-negotiable: - Work ONLY inside the current repository working directory. - Green/Blue lane: file edits and local git add/commit only. Never push, never touch the network, never run destructive commands. +- Tool profile for this run: {tool_profile} (lane={lane}). - Bounded effort: complete the single task below, commit with a clear message, then stop. If the task cannot be completed, commit nothing and say why in your final output. @@ -45,6 +49,10 @@ class RunResult: persona_source: str session_output: str reason: str = "" + tool_profile: str = "" + budget_tokens: int | None = None + tokens_spent: int | None = None + execution_time_s: float = 0.0 def _git(repo: Path, *args: str) -> str: @@ -59,11 +67,49 @@ def _git(repo: Path, *args: str) -> str: return result.stdout.strip() -def run_task(spec: TaskSpec, adapter=None, report_to_hub: bool = True) -> RunResult: +def run_task( + spec: TaskSpec, + adapter=None, + report_to_hub: bool = True, + write_metrics: bool = True, +) -> RunResult: + try: + profile_name, budget_tokens, lane = resolve_run_policy( + spec.target_repo, spec.agent + ) + profile = get_profile(profile_name) + except UnknownToolProfileError as exc: + return RunResult( + ok=False, + committed=False, + head_before="", + head_after="", + persona_source="none", + session_output="", + reason=f"refused: {exc}", + tool_profile="", + budget_tokens=None, + ) + except Exception as exc: + return RunResult( + ok=False, + committed=False, + head_before="", + head_after="", + persona_source="none", + session_output="", + reason=f"manifest resolution failed: {exc}", + tool_profile="", + budget_tokens=None, + ) + if adapter is None: from agent_harness.adapter import AgenticClaudeCodeAdapter - adapter = AgenticClaudeCodeAdapter(workdir=spec.target_repo) + adapter = AgenticClaudeCodeAdapter( + workdir=spec.target_repo, + tool_profile=profile, + ) head_before = _git(spec.target_repo, "rev-parse", "HEAD") persona, persona_source = load_persona_bundle(spec.agent, spec.target_repo) @@ -71,20 +117,30 @@ def run_task(spec: TaskSpec, adapter=None, report_to_hub: bool = True) -> RunRes persona=persona or "(no persona bundle available for this run)", title=spec.title, description=spec.description, + tool_profile=profile.name, + lane=lane or profile.lane, ) - from llm_connect.models import RunConfig + from llm_connect.models import BudgetTracker, RunConfig - config = RunConfig(timeout_seconds=spec.timeout_seconds, skip_if_exists=False) + budget_tracker = BudgetTracker(total=budget_tokens) if budget_tokens else None + config = RunConfig( + timeout_seconds=spec.timeout_seconds, + skip_if_exists=False, + budget_tracker=budget_tracker, + ) + + started = time.monotonic() try: response = adapter.execute_prompt(prompt, config) session_output = response.content session_ok = True reason = "" - except Exception as exc: # adapter failures must still be reported + except Exception as exc: # adapter / budget failures must still be reported session_output = "" session_ok = False reason = f"session failed: {exc}" + execution_time_s = time.monotonic() - started head_after = _git(spec.target_repo, "rev-parse", "HEAD") committed = head_after != head_before @@ -92,6 +148,8 @@ def run_task(spec: TaskSpec, adapter=None, report_to_hub: bool = True) -> RunRes if session_ok and not committed: reason = "session completed without committing" + tokens_spent = budget_tracker.spent if budget_tracker is not None else None + result = RunResult( ok=ok, committed=committed, @@ -100,8 +158,33 @@ def run_task(spec: TaskSpec, adapter=None, report_to_hub: bool = True) -> RunRes persona_source=persona_source, session_output=session_output, reason=reason, + tool_profile=profile.name, + budget_tokens=budget_tokens, + tokens_spent=tokens_spent, + execution_time_s=execution_time_s, ) + if write_metrics: + try: + metrics.record_execution( + spec.target_repo, + spec.agent, + success=ok, + execution_time_s=execution_time_s, + tokens=tokens_spent, + committed=committed, + head_after=head_after, + reason=reason or None, + metadata={ + "task_title": spec.title, + "tool_profile": profile.name, + "labels": list(spec.labels), + "completion_event_type": spec.completion_event_type, + }, + ) + except OSError: + pass # metrics must not block run completion reporting + if report_to_hub: detail = { "repo": spec.target_repo.name, @@ -113,6 +196,10 @@ def run_task(spec: TaskSpec, adapter=None, report_to_hub: bool = True) -> RunRes "head_after": head_after, "ok": ok, "reason": reason, + "tool_profile": profile.name, + "budget_tokens": budget_tokens, + "tokens_spent": tokens_spent, + "execution_time_s": round(execution_time_s, 3), } hub.post_progress_event( summary=f"executor run: {spec.title} ({'ok' if ok else 'failed'})", @@ -120,6 +207,15 @@ def run_task(spec: TaskSpec, adapter=None, report_to_hub: bool = True) -> RunRes detail=detail, task_id=spec.hub_task_id, ) + if tokens_spent is not None or budget_tokens is not None: + hub.post_token_event( + repo=spec.target_repo.name, + tokens=tokens_spent or 0, + budget=budget_tokens, + agent=spec.agent, + ok=ok, + detail={"task_title": spec.title, "tool_profile": profile.name}, + ) if ok and spec.hub_task_id: hub.close_task(spec.hub_task_id) diff --git a/docs/architecture.md b/docs/architecture.md index fb461a1..b1402d4 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -32,25 +32,27 @@ | Component | File | Today | Target | |---|---|---|---| +| Instance manifest | `manifest.py` | extends `.kaizen/schedule.yml`; `validate` CLI | unchanged contract; tenant onboarding | +| Tool profiles | `profiles.py` | `green-commit-only`, `blue-mail-triage` registry | additional named profiles as needed | | Task intake | `taskspec.py` | JSON task-spec file | poll issue-core sink / TaskExecutorWorkflow | | Persona | `persona.py` | `kaizen-agentic schedule prepare` (ADR-005) | unchanged, plus phase-memory profile hook | -| Session | `adapter.py` | `AgenticClaudeCodeAdapter` (llm-connect subclass, cwd-pinned, hard allow-list) | + hosted adapters; tool profiles (below) | -| Orchestration | `runner.py` | HEAD snapshot → session → commit check → report | + budget enforcement, kaizen metrics write | +| Session | `adapter.py` | `AgenticClaudeCodeAdapter` (cwd-pinned, profile allow-list) | + hosted adapters | +| Orchestration | `runner.py` | profile/budget → session → commit → metrics → hub | activity-core intake | +| Metrics | `metrics.py` | ADR-004 `.kaizen/metrics` write per run | correlate with Helix fleet metrics | | Mail lane | `mailscan.py` | deterministic credentialed pre-step outside the session | pattern generalizes to other credentialed pre-steps | -| Hub reporting | `hub.py` | REST progress event + task close | unchanged | +| Hub reporting | `hub.py` | REST progress event + task close + token event | unchanged | ## Contracts -**Instance manifest** (in the consuming repo; `.kaizen/schedule.yml` today, -extended per-agent fields as the manifest spec lands in HARNESS-WP-0001): -blueprint name, cadence, `enabled`, plus target additions: `lane` -(green/blue), `tool_profile` (named, defined here), `budget` -(tokens/run), `harness` (pinned major version). +**Instance manifest** — `.kaizen/schedule.yml` (ADR-005 base + harness +extensions). Full contract: [instance-manifest.md](instance-manifest.md). +Fields: blueprint, cadence, `enabled`, `lane` (green/blue), +`tool_profile` (named), `budget` (tokens/run), `harness` (pinned major). -**Tool profiles** are named allow-lists defined centrally in the harness, -referenced by name from manifests — instances never enumerate tools. -Seed profile `green-commit-only` = Read/Write/Edit/Glob/Grep + local git -add/commit/status/log/diff. No push, no network, no arbitrary shell. +**Tool profiles** are named allow-lists defined centrally in the harness +(`profiles.py`), referenced by name from manifests — instances never +enumerate tools. Seed: `green-commit-only`, `blue-mail-triage`. No push, +no network, no arbitrary shell. Unknown profile = refuse to run. **Completion events** are the idempotence currency: each run posts a progress event (e.g. `binky_daily_brief`, `binky_mail_intake`) with diff --git a/docs/instance-manifest.md b/docs/instance-manifest.md new file mode 100644 index 0000000..fc0df58 --- /dev/null +++ b/docs/instance-manifest.md @@ -0,0 +1,105 @@ +# Instance manifest + +> Contract for declarative agent instances in consuming repos. +> Companion to ADR-001, ADR-005 (kaizen-agentic), and HARNESS-WP-0001-T01. + +## Location + +``` +/.kaizen/schedule.yml +``` + +**Same file as ADR-005.** kaizen-agentic owns the base schedule keys; +agent-harness owns the extension keys below. A sibling file is reserved +only if kaizen owners later prefer hard separation — until then, one +manifest keeps cadence and runtime policy colocated. + +Validate: + +```bash +kaizen-agentic schedule validate --target # base ADR-005 keys +agent-harness validate --target # base + harness extensions +agent-harness validate --target --strict # require harness fields on enabled agents +``` + +## Schema + +| Key | Owner | Required | Type | Notes | +|-----|-------|----------|------|-------| +| `version` | kaizen | yes | string | Must be `"1"` | +| `timezone` | kaizen | no | string | IANA tz | +| `harness` | harness | no\* | int | Pinned harness **major** for the repo | +| `agents` | both | yes | mapping | `name → settings` | +| `agents..cadence` | kaizen | yes | enum | `daily` \| `weekly` \| `monthly` | +| `agents..cron` | kaizen | no | string | 5-field cron override | +| `agents..enabled` | kaizen | no | bool | Default `true` | +| `agents..blueprint` | harness | no | string | Defaults to agent name (kaizen blueprint) | +| `agents..lane` | harness | no\* | enum | `green` \| `blue` | +| `agents..tool_profile` | harness | no\* | string | Named profile in the harness registry | +| `agents..budget` | harness | no | int | Token cap per run (positive) | +| `agents..harness` | harness | no | int | Per-agent major pin; overrides top-level | + +\* Required for enabled agents under `agent-harness validate --strict`. + +## Example (tenant-ready) + +```yaml +# .kaizen/schedule.yml — ADR-005 + agent-harness extensions +version: "1" +timezone: Europe/Berlin +harness: 0 +agents: + coach: + cadence: daily + cron: "0 7 * * 1-5" + enabled: true + lane: green + tool_profile: green-commit-only + budget: 80000 + mail-triage: + cadence: weekly + cron: "0 8 * * 1" + enabled: true + blueprint: coach + lane: blue + tool_profile: blue-mail-triage + budget: 40000 + review-prep: + cadence: weekly + cron: "0 9 * * 5" + enabled: true + lane: green + tool_profile: green-commit-only + budget: 60000 +``` + +## Tool profiles + +Defined **only** in agent-harness (see `agent_harness/profiles.py`). +Manifests reference them by name; unknown names refuse to run. + +| Name | Lane | Session tools | +|------|------|---------------| +| `green-commit-only` | green | Read/Write/Edit/Glob/Grep + local git add/commit/status/log/diff (+ date, ls) | +| `blue-mail-triage` | blue | Same session tools; credentialed IMAP scan is a deterministic pre-step outside the session | + +No push, no network, no arbitrary shell in either profile. + +## Budget + +`budget` is tokens per run. The runner wires llm-connect `BudgetTracker` +when set; exhaustion refuses or truncates the run and is reported to the +State Hub (token events feed the Token Cost dashboard). + +## Harness pin + +Instances pin a harness **major**. This runtime implements major `0` +(package `0.x`). A mismatched pin fails `validate` so upgrades are +deliberate. + +## Relationship to task files + +Local development may still use JSON task files (`agent-harness run +--task-file …`). When the target repo has a matching agent entry, the +runner resolves `tool_profile`, `budget`, and `lane` from the manifest; +otherwise it defaults to `green-commit-only` with no budget cap. diff --git a/examples/schedule.harness.yml b/examples/schedule.harness.yml new file mode 100644 index 0000000..d02de89 --- /dev/null +++ b/examples/schedule.harness.yml @@ -0,0 +1,30 @@ +# Example instance manifest (ADR-005 + agent-harness extensions). +# Copy keys into a consuming repo's .kaizen/schedule.yml. +# Validate with: +# agent-harness validate --target --strict +version: "1" +timezone: Europe/Berlin +harness: 0 +agents: + coach: + cadence: daily + cron: "0 7 * * 1-5" + enabled: true + lane: green + tool_profile: green-commit-only + budget: 80000 + mail-triage: + cadence: weekly + cron: "0 8 * * 1" + enabled: true + blueprint: coach + lane: blue + tool_profile: blue-mail-triage + budget: 40000 + review-prep: + cadence: weekly + cron: "0 9 * * 5" + enabled: true + lane: green + tool_profile: green-commit-only + budget: 60000 diff --git a/pyproject.toml b/pyproject.toml index c2e80d9..0f60ed4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,6 +5,7 @@ description = "Thin executor worker: consumes emitted activity-core tasks and ex requires-python = ">=3.11" dependencies = [ "httpx>=0.27", + "PyYAML>=6.0", ] [project.scripts] diff --git a/tests/test_manifest.py b/tests/test_manifest.py new file mode 100644 index 0000000..7bae098 --- /dev/null +++ b/tests/test_manifest.py @@ -0,0 +1,215 @@ +from __future__ import annotations + +from pathlib import Path + +import pytest +import yaml + +from agent_harness.cli import main +from agent_harness.manifest import ( + HARNESS_MAJOR, + ManifestError, + load_manifest, + parse_manifest, + resolve_run_policy, + validate_manifest, +) +from agent_harness.profiles import UnknownToolProfileError, get_profile, list_profiles + + +def _write_manifest(tmp_path: Path, data: dict) -> Path: + kaizen = tmp_path / ".kaizen" + kaizen.mkdir() + path = kaizen / "schedule.yml" + path.write_text(yaml.safe_dump(data, sort_keys=False), encoding="utf-8") + return path + + +def test_parse_and_validate_adr005_only() -> None: + manifest = parse_manifest( + { + "version": "1", + "timezone": "Europe/Berlin", + "agents": { + "coach": {"cadence": "weekly", "enabled": True}, + }, + } + ) + assert validate_manifest(manifest) == [] + assert validate_manifest(manifest, require_harness_fields=True) + + +def test_validate_harness_extensions() -> None: + manifest = parse_manifest( + { + "version": "1", + "harness": HARNESS_MAJOR, + "agents": { + "coach": { + "cadence": "daily", + "enabled": True, + "lane": "green", + "tool_profile": "green-commit-only", + "budget": 50000, + }, + "mail": { + "cadence": "weekly", + "lane": "blue", + "tool_profile": "blue-mail-triage", + "budget": 10000, + }, + }, + } + ) + assert validate_manifest(manifest) == [] + assert validate_manifest(manifest, require_harness_fields=True) == [] + + +def test_unknown_tool_profile_errors() -> None: + manifest = parse_manifest( + { + "version": "1", + "agents": { + "coach": { + "cadence": "daily", + "tool_profile": "does-not-exist", + } + }, + } + ) + errors = validate_manifest(manifest) + assert any("unknown tool_profile" in e for e in errors) + + +def test_lane_profile_mismatch() -> None: + manifest = parse_manifest( + { + "version": "1", + "agents": { + "coach": { + "cadence": "daily", + "lane": "blue", + "tool_profile": "green-commit-only", + } + }, + } + ) + errors = validate_manifest(manifest) + assert any("does not match tool_profile" in e for e in errors) + + +def test_budget_must_be_positive() -> None: + with pytest.raises(ManifestError, match="budget"): + parse_manifest( + { + "version": "1", + "agents": {"coach": {"cadence": "daily", "budget": 0}}, + } + ) + + +def test_harness_major_mismatch() -> None: + manifest = parse_manifest( + { + "version": "1", + "harness": HARNESS_MAJOR + 1, + "agents": {"coach": {"cadence": "daily"}}, + } + ) + errors = validate_manifest(manifest) + assert any("does not match this runtime" in e for e in errors) + + +def test_resolve_run_policy_defaults(tmp_path: Path) -> None: + profile, budget, lane = resolve_run_policy(tmp_path, "coach") + assert profile == "green-commit-only" + assert budget is None + assert lane is None + + +def test_resolve_run_policy_from_manifest(tmp_path: Path) -> None: + _write_manifest( + tmp_path, + { + "version": "1", + "harness": HARNESS_MAJOR, + "agents": { + "coach": { + "cadence": "daily", + "tool_profile": "blue-mail-triage", + "lane": "blue", + "budget": 12000, + } + }, + }, + ) + profile, budget, lane = resolve_run_policy(tmp_path, "coach") + assert profile == "blue-mail-triage" + assert budget == 12000 + assert lane == "blue" + + +def test_resolve_unknown_profile_refuses(tmp_path: Path) -> None: + _write_manifest( + tmp_path, + { + "version": "1", + "agents": { + "coach": { + "cadence": "daily", + "tool_profile": "nope", + } + }, + }, + ) + with pytest.raises(UnknownToolProfileError): + resolve_run_policy(tmp_path, "coach") + + +def test_load_manifest_missing(tmp_path: Path) -> None: + with pytest.raises(ManifestError, match="not found"): + load_manifest(tmp_path / ".kaizen" / "schedule.yml") + + +def test_profiles_registry() -> None: + names = {p.name for p in list_profiles()} + assert names == {"green-commit-only", "blue-mail-triage"} + green = get_profile("green-commit-only") + assert "git commit" in green.allowed_tools + assert "git push" not in green.allowed_tools + + +def test_cli_validate_ok(tmp_path: Path) -> None: + _write_manifest( + tmp_path, + { + "version": "1", + "harness": HARNESS_MAJOR, + "agents": { + "coach": { + "cadence": "daily", + "lane": "green", + "tool_profile": "green-commit-only", + "budget": 1000, + } + }, + }, + ) + assert main(["validate", "--target", str(tmp_path), "--strict"]) == 0 + + +def test_cli_validate_fails_unknown_profile(tmp_path: Path) -> None: + _write_manifest( + tmp_path, + { + "version": "1", + "agents": { + "coach": {"cadence": "daily", "tool_profile": "missing"} + }, + }, + ) + assert main(["validate", "--target", str(tmp_path)]) == 1 + + +def test_cli_profiles() -> None: + assert main(["profiles"]) == 0 diff --git a/tests/test_metrics.py b/tests/test_metrics.py new file mode 100644 index 0000000..171849e --- /dev/null +++ b/tests/test_metrics.py @@ -0,0 +1,51 @@ +from __future__ import annotations + +import json +from pathlib import Path + +from agent_harness.metrics import record_execution, regenerate_summary + + +def test_record_execution_writes_jsonl_and_summary(tmp_path: Path) -> None: + path = record_execution( + tmp_path, + "coach", + success=True, + execution_time_s=12.5, + tokens=100, + committed=True, + head_after="abc123", + reason=None, + metadata={"task_title": "hello"}, + ) + assert path.is_file() + lines = path.read_text(encoding="utf-8").strip().splitlines() + assert len(lines) == 1 + rec = json.loads(lines[0]) + assert rec["agent"] == "coach" + assert rec["success"] is True + assert rec["tokens"] == 100 + assert rec["harness"] == "agent-harness" + assert rec["metadata"]["task_title"] == "hello" + + summary = json.loads( + (tmp_path / ".kaizen" / "metrics" / "coach" / "summary.json").read_text() + ) + assert summary["execution_count"] == 1 + assert summary["success_rate"] == 1.0 + assert summary["avg_execution_time_s"] == 12.5 + + +def test_summary_aggregates_multiple(tmp_path: Path) -> None: + record_execution(tmp_path, "coach", success=True, execution_time_s=10) + record_execution(tmp_path, "coach", success=False, execution_time_s=20) + summary = json.loads( + (tmp_path / ".kaizen" / "metrics" / "coach" / "summary.json").read_text() + ) + assert summary["execution_count"] == 2 + assert summary["success_rate"] == 0.5 + assert summary["avg_execution_time_s"] == 15.0 + + +def test_regenerate_summary_empty() -> None: + assert regenerate_summary("x", [])["execution_count"] == 0 diff --git a/tests/test_runner.py b/tests/test_runner.py index 04ea53e..89774eb 100644 --- a/tests/test_runner.py +++ b/tests/test_runner.py @@ -1,10 +1,13 @@ from __future__ import annotations +import json import subprocess from pathlib import Path import pytest +import yaml +from agent_harness.manifest import HARNESS_MAJOR from agent_harness.runner import RunResult, run_task from agent_harness.taskspec import TaskSpec, TaskSpecError @@ -23,15 +26,26 @@ def _make_repo(tmp_path: Path) -> Path: return repo +def _write_manifest(repo: Path, agents: dict) -> None: + kaizen = repo / ".kaizen" + kaizen.mkdir(exist_ok=True) + data = {"version": "1", "harness": HARNESS_MAJOR, "agents": agents} + (kaizen / "schedule.yml").write_text( + yaml.safe_dump(data, sort_keys=False), encoding="utf-8" + ) + + class CommittingAdapter: """Fake adapter that simulates a session which commits.""" def __init__(self, repo: Path): self.repo = repo self.prompts: list[str] = [] + self.configs: list = [] def execute_prompt(self, prompt, config): self.prompts.append(prompt) + self.configs.append(config) (self.repo / "HELLO.md").write_text("hello\n") subprocess.run(["git", "add", "."], cwd=self.repo, check=True) subprocess.run( @@ -51,34 +65,140 @@ class IdleAdapter: return LLMResponse(content="nothing to do", model="fake", usage={}, finish_reason="stop") -def _spec(repo: Path) -> TaskSpec: - return TaskSpec(title="write hello", description="create HELLO.md", target_repo=repo) +class BudgetBlowingAdapter: + def execute_prompt(self, prompt, config): + from llm_connect.exceptions import LLMBudgetExceededError + + if config.budget_tracker is not None: + # Simulate preflight/exhaustion the way adapters do. + config.budget_tracker.consume(config.budget_tracker.total) + config.budget_tracker.consume(1) + raise LLMBudgetExceededError( + "Token budget exceeded", + total=1, + spent=1, + requested=1, + ) + + +def _spec(repo: Path, agent: str = "coach") -> TaskSpec: + return TaskSpec( + title="write hello", + description="create HELLO.md", + target_repo=repo, + agent=agent, + ) def test_run_task_success_when_session_commits(tmp_path) -> None: repo = _make_repo(tmp_path) adapter = CommittingAdapter(repo) - result = run_task(_spec(repo), adapter=adapter, report_to_hub=False) + result = run_task( + _spec(repo), adapter=adapter, report_to_hub=False, write_metrics=True + ) assert isinstance(result, RunResult) assert result.ok is True assert result.committed is True assert result.head_before != result.head_after + assert result.tool_profile == "green-commit-only" assert "write hello" in adapter.prompts[0] assert "Never push" in adapter.prompts[0] + assert "Tool profile for this run: green-commit-only" in adapter.prompts[0] + + metrics_path = repo / ".kaizen" / "metrics" / "coach" / "executions.jsonl" + assert metrics_path.is_file() + rec = json.loads(metrics_path.read_text().strip().splitlines()[-1]) + assert rec["success"] is True + assert rec["committed"] is True def test_run_task_fails_without_commit(tmp_path) -> None: repo = _make_repo(tmp_path) - result = run_task(_spec(repo), adapter=IdleAdapter(), report_to_hub=False) + result = run_task( + _spec(repo), adapter=IdleAdapter(), report_to_hub=False, write_metrics=False + ) assert result.ok is False assert result.committed is False assert result.reason == "session completed without committing" +def test_run_task_refuses_unknown_tool_profile(tmp_path) -> None: + repo = _make_repo(tmp_path) + _write_manifest( + repo, + { + "coach": { + "cadence": "daily", + "tool_profile": "not-a-real-profile", + } + }, + ) + + result = run_task( + _spec(repo), adapter=IdleAdapter(), report_to_hub=False, write_metrics=False + ) + + assert result.ok is False + assert result.reason.startswith("refused:") + assert "not-a-real-profile" in result.reason + + +def test_run_task_resolves_manifest_profile_and_budget(tmp_path) -> None: + repo = _make_repo(tmp_path) + _write_manifest( + repo, + { + "coach": { + "cadence": "daily", + "lane": "blue", + "tool_profile": "blue-mail-triage", + "budget": 25000, + } + }, + ) + adapter = CommittingAdapter(repo) + + result = run_task( + _spec(repo), adapter=adapter, report_to_hub=False, write_metrics=False + ) + + assert result.ok is True + assert result.tool_profile == "blue-mail-triage" + assert result.budget_tokens == 25000 + assert adapter.configs[0].budget_tracker is not None + assert adapter.configs[0].budget_tracker.total == 25000 + assert "blue-mail-triage" in adapter.prompts[0] + + +def test_run_task_budget_exhaustion_fails(tmp_path) -> None: + repo = _make_repo(tmp_path) + _write_manifest( + repo, + { + "coach": { + "cadence": "daily", + "tool_profile": "green-commit-only", + "budget": 10, + } + }, + ) + + result = run_task( + _spec(repo), + adapter=BudgetBlowingAdapter(), + report_to_hub=False, + write_metrics=False, + ) + + assert result.ok is False + assert "session failed" in result.reason + assert result.budget_tokens == 10 + + def test_taskspec_rejects_non_repo(tmp_path) -> None: spec_file = tmp_path / "task.json" spec_file.write_text( diff --git a/workplans/HARNESS-WP-0001-harness-foundation.md b/workplans/HARNESS-WP-0001-harness-foundation.md index 5a0b60b..515ec16 100644 --- a/workplans/HARNESS-WP-0001-harness-foundation.md +++ b/workplans/HARNESS-WP-0001-harness-foundation.md @@ -21,7 +21,7 @@ sibling manifest if they prefer separation). ```task id: HARNESS-WP-0001-T01 -status: todo +status: done priority: high state_hub_task_id: "ee8f6f16-8e34-4961-b765-8ebad8622cda" ``` @@ -35,7 +35,7 @@ profiles by name; the runner resolves and enforces them. Unknown profile ```task id: HARNESS-WP-0001-T02 -status: todo +status: done priority: high state_hub_task_id: "a043a228-e306-49af-9500-c68841b4dc40" ``` @@ -63,7 +63,7 @@ observes harness-run agents. Follow kaizen-agentic's metrics conventions ```task id: HARNESS-WP-0001-T04 -status: todo +status: done priority: medium state_hub_task_id: "1496a3e9-4391-4497-8b55-782ea0fc6525" ``` @@ -76,7 +76,7 @@ events feed the Token Cost dashboard). ```task id: HARNESS-WP-0001-T05 -status: todo +status: done priority: medium state_hub_task_id: "9a388b80-47ac-41bd-a9ea-776f2cbd37e5" ```