Land HARNESS-WP-0001 T01/T02/T04/T05: extend ADR-005 schedule.yml with harness fields, named tool-profile registry, ADR-004 metrics writes, and BudgetTracker wiring. CLI gains validate/profiles; task-file path kept.
51 lines
1.6 KiB
Python
51 lines
1.6 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
from agent_harness.metrics import record_execution, regenerate_summary
|
|
|
|
|
|
def test_record_execution_writes_jsonl_and_summary(tmp_path: Path) -> None:
|
|
path = record_execution(
|
|
tmp_path,
|
|
"coach",
|
|
success=True,
|
|
execution_time_s=12.5,
|
|
tokens=100,
|
|
committed=True,
|
|
head_after="abc123",
|
|
reason=None,
|
|
metadata={"task_title": "hello"},
|
|
)
|
|
assert path.is_file()
|
|
lines = path.read_text(encoding="utf-8").strip().splitlines()
|
|
assert len(lines) == 1
|
|
rec = json.loads(lines[0])
|
|
assert rec["agent"] == "coach"
|
|
assert rec["success"] is True
|
|
assert rec["tokens"] == 100
|
|
assert rec["harness"] == "agent-harness"
|
|
assert rec["metadata"]["task_title"] == "hello"
|
|
|
|
summary = json.loads(
|
|
(tmp_path / ".kaizen" / "metrics" / "coach" / "summary.json").read_text()
|
|
)
|
|
assert summary["execution_count"] == 1
|
|
assert summary["success_rate"] == 1.0
|
|
assert summary["avg_execution_time_s"] == 12.5
|
|
|
|
|
|
def test_summary_aggregates_multiple(tmp_path: Path) -> None:
|
|
record_execution(tmp_path, "coach", success=True, execution_time_s=10)
|
|
record_execution(tmp_path, "coach", success=False, execution_time_s=20)
|
|
summary = json.loads(
|
|
(tmp_path / ".kaizen" / "metrics" / "coach" / "summary.json").read_text()
|
|
)
|
|
assert summary["execution_count"] == 2
|
|
assert summary["success_rate"] == 0.5
|
|
assert summary["avg_execution_time_s"] == 15.0
|
|
|
|
|
|
def test_regenerate_summary_empty() -> None:
|
|
assert regenerate_summary("x", [])["execution_count"] == 0
|