rein-aharness/tests/test_metrics.py
tegwick 4144eba160 feat: instance manifest, tool profiles, metrics, and budget enforcement
Land HARNESS-WP-0001 T01/T02/T04/T05: extend ADR-005 schedule.yml with
harness fields, named tool-profile registry, ADR-004 metrics writes, and
BudgetTracker wiring. CLI gains validate/profiles; task-file path kept.
2026-07-17 23:49:03 +02:00

51 lines
1.6 KiB
Python

from __future__ import annotations
import json
from pathlib import Path
from agent_harness.metrics import record_execution, regenerate_summary
def test_record_execution_writes_jsonl_and_summary(tmp_path: Path) -> None:
path = record_execution(
tmp_path,
"coach",
success=True,
execution_time_s=12.5,
tokens=100,
committed=True,
head_after="abc123",
reason=None,
metadata={"task_title": "hello"},
)
assert path.is_file()
lines = path.read_text(encoding="utf-8").strip().splitlines()
assert len(lines) == 1
rec = json.loads(lines[0])
assert rec["agent"] == "coach"
assert rec["success"] is True
assert rec["tokens"] == 100
assert rec["harness"] == "agent-harness"
assert rec["metadata"]["task_title"] == "hello"
summary = json.loads(
(tmp_path / ".kaizen" / "metrics" / "coach" / "summary.json").read_text()
)
assert summary["execution_count"] == 1
assert summary["success_rate"] == 1.0
assert summary["avg_execution_time_s"] == 12.5
def test_summary_aggregates_multiple(tmp_path: Path) -> None:
record_execution(tmp_path, "coach", success=True, execution_time_s=10)
record_execution(tmp_path, "coach", success=False, execution_time_s=20)
summary = json.loads(
(tmp_path / ".kaizen" / "metrics" / "coach" / "summary.json").read_text()
)
assert summary["execution_count"] == 2
assert summary["success_rate"] == 0.5
assert summary["avg_execution_time_s"] == 15.0
def test_regenerate_summary_empty() -> None:
assert regenerate_summary("x", [])["execution_count"] == 0