Implement canon conformance and maintenance optimizations
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a06e82-3e08-7042-a79d-438ac6eed8db
This commit is contained in:
parent
a2e7f22d8d
commit
b081d39da1
64 changed files with 4491 additions and 373 deletions
108
src/info_tech_canon/maintenance.py
Normal file
108
src/info_tech_canon/maintenance.py
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
"""Reproducible projections, portable contract exports, and read measurements."""
|
||||
|
||||
from collections import Counter
|
||||
from dataclasses import replace
|
||||
from datetime import datetime, timezone
|
||||
import hashlib
|
||||
import io
|
||||
import json
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
import statistics
|
||||
import subprocess
|
||||
import tarfile
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
from . import generation
|
||||
|
||||
|
||||
def scope_inventory(context) -> dict:
|
||||
return {"artifact_count": len(context.infospace.artifacts),
|
||||
"kinds": dict(sorted(Counter(a.kind for a in context.infospace.artifacts).items())),
|
||||
"standards": sorted(a.id for a in context.infospace.artifacts if a.kind == "standard"),
|
||||
"models": sorted(a.id for a in context.infospace.artifacts if a.kind == "model")}
|
||||
|
||||
|
||||
def check_generated(context) -> dict:
|
||||
"""Render only into a disposable copy; never repair the source as a check."""
|
||||
with tempfile.TemporaryDirectory(prefix="canon-freshness-") as temporary:
|
||||
target = Path(temporary) / "infospace"
|
||||
shutil.copytree(context.infospace_root, target)
|
||||
copied = replace(context, infospace_root=target)
|
||||
files = {}
|
||||
for render in (generation.generate_indexes, generation.generate_tree,
|
||||
generation.generate_agent_briefs):
|
||||
for item in render(copied)["files"]:
|
||||
path = Path(item["path"])
|
||||
files[str(path.relative_to(target))] = path.read_bytes()
|
||||
stale = []
|
||||
for relative, expected in files.items():
|
||||
original = context.infospace_root / relative
|
||||
if not original.is_file() or original.read_bytes() != expected:
|
||||
stale.append(relative)
|
||||
return {"ok": not stale, "checked": len(files), "stale": sorted(stale)}
|
||||
|
||||
|
||||
def source_evidence(root: Path) -> dict:
|
||||
digest = hashlib.sha256()
|
||||
# Include source and generated corpus; exclude reports to avoid self-hashing.
|
||||
for path in sorted(root.rglob("*")):
|
||||
if path.is_file() and "validation" not in path.relative_to(root).parts:
|
||||
digest.update(str(path.relative_to(root)).encode() + b"\0" + path.read_bytes() + b"\0")
|
||||
def git(*args):
|
||||
try:
|
||||
result = subprocess.run(["git", "-C", str(root.parent), *args],
|
||||
capture_output=True, text=True, check=False)
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
return result.stdout.strip() if result.returncode == 0 else None
|
||||
status = git("status", "--porcelain")
|
||||
return {"generated_at": datetime.now(timezone.utc).isoformat(),
|
||||
"revision": git("rev-parse", "HEAD"),
|
||||
"dirty": bool(status) if status is not None else None,
|
||||
"corpus_sha256": digest.hexdigest(),
|
||||
"digest_scope": "all infospace files except validation reports"}
|
||||
|
||||
|
||||
def export_emission_bundle(root: Path, destination: Path) -> dict:
|
||||
"""Content-addressed deterministic tar; a changed contract gets a new name."""
|
||||
paths = ["schemas/emission-cadence.schema.yaml",
|
||||
"standards/emission-cadence/InfoTechCanonEmissionCadenceStandard.md",
|
||||
"standards/emission-cadence/examples/qonto-assistant.yaml"]
|
||||
files = {path: (root / path).read_bytes() for path in paths}
|
||||
manifest = {"contract": "emission-cadence/0.1", "status": "draft",
|
||||
"files": {name: hashlib.sha256(data).hexdigest() for name, data in files.items()},
|
||||
"semantic_checks": ["unique source_id in sources"],
|
||||
"adoption_evidence": "Two independent source-owned implementations required; example does not count."}
|
||||
files["manifest.json"] = (json.dumps(manifest, sort_keys=True, indent=2) + "\n").encode()
|
||||
output = io.BytesIO()
|
||||
with tarfile.open(fileobj=output, mode="w", format=tarfile.USTAR_FORMAT) as archive:
|
||||
for name, data in sorted(files.items()):
|
||||
info = tarfile.TarInfo(name)
|
||||
info.size = len(data)
|
||||
info.mode = 0o644
|
||||
archive.addfile(info, io.BytesIO(data))
|
||||
data = output.getvalue()
|
||||
digest = hashlib.sha256(data).hexdigest()
|
||||
destination.mkdir(parents=True, exist_ok=True)
|
||||
path = destination / f"emission-cadence-0.1-{digest}.tar"
|
||||
try:
|
||||
with path.open("xb") as stream:
|
||||
stream.write(data)
|
||||
except FileExistsError:
|
||||
if path.read_bytes() != data:
|
||||
raise ValueError(f"Refusing to overwrite different bundle contents: {path}")
|
||||
return {"ok": True, "path": str(path), "sha256": digest, "manifest": manifest}
|
||||
|
||||
|
||||
def benchmark_reads(root: Path, runs: int = 10) -> dict:
|
||||
from .service import inspect_canon
|
||||
samples = []
|
||||
for _ in range(runs):
|
||||
started = time.perf_counter()
|
||||
inspect_canon(root)
|
||||
samples.append((time.perf_counter() - started) * 1000)
|
||||
return {"ok": True, "operation": "inspect", "runs": runs,
|
||||
"median_ms": statistics.median(samples), "max_ms": max(samples),
|
||||
"cache": "none; benchmark does not establish a production latency requirement"}
|
||||
Loading…
Add table
Add a link
Reference in a new issue