audit-core/audit_core/sqlite_backend.py
tegwick eb649dd747
All checks were successful
CI Smoke / host-smoke (push) Successful in 0s
CI Smoke / container-smoke (push) Successful in 1s
Route ingestion through the backend contract; fix error semantics
AUDIT-WP-0004 T01, T02, T07.

T01 - ingestion wrote to SQLite directly and never called the AuditBackend
contract, so a 202 meant a row existed rather than that a backend with a
declared retention policy had accepted the event. Adds IdempotentAuditBackend
to the contract: duplicate detection lives inside the backend so custody and
idempotency state share a transaction and cannot diverge. SQLiteAuditBackend
implements it with WAL, synchronous=FULL and a busy timeout. Ingestion now
refuses any backend declaring durable=False, so the development file backend
cannot silently become the production sink.

The atomicity claim was tested rather than asserted, and the first attempt
failed: with a single shared connection, 16 racing submissions of one event
told two callers they were first. Storage was correct but the response was
not. Fixed with per-thread connections and BEGIN IMMEDIATE around the
insert/read pair, and locked in by a test.

T02 - storage errors previously escaped the handler with start_response never
called, and the auth check sat outside the try block so a non-ASCII
Authorization header crashed the request. Adds a catch-all, maps conflict to
409, backend unavailability to 503 and unexpected faults to 500, and
documents the full response contract with the retry semantics each status
implies, since senders key their behaviour off it.

T07 - ingestion tests 2 -> 23, suite 15 -> 36. accepted_at is now UTC rather
than local time, and naive timestamps are rejected instead of silently
assumed.

Remaining in WP-0004: T03 tenant/source binding, T04 redaction policy,
T05 operator read surface, T06 production serving layer.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-10 14:30:20 +02:00

177 lines
6.2 KiB
Python

"""Durable SQLite audit backend.
Implements the idempotent backend contract for single-node deployments and for
development. Production custody moves to PostgreSQL under AUDIT-WP-0005; this
backend stays the development and test implementation and defines the
behaviour the Postgres backend must match.
"""
from __future__ import annotations
import json
import sqlite3
import threading
from datetime import datetime, timezone
from audit_core.interface import (
AcceptResult,
AuditEvent,
BackendUnavailableError,
EventConflictError,
EventValidationError,
RetentionPolicy,
validate_event,
)
_SCHEMA = """
CREATE TABLE IF NOT EXISTS events (
event_id TEXT PRIMARY KEY,
payload_hash TEXT NOT NULL,
accepted_at TEXT NOT NULL,
correlation_id TEXT,
tenant TEXT NOT NULL,
record TEXT NOT NULL
);
CREATE INDEX IF NOT EXISTS events_correlation_idx ON events (correlation_id);
CREATE INDEX IF NOT EXISTS events_tenant_idx ON events (tenant);
"""
class SQLiteAuditBackend:
"""Store audit events in SQLite with idempotent accept semantics.
Configured for durability rather than speed: WAL journalling, ``synchronous
= FULL`` so an acknowledged write has reached disk, and a busy timeout so
concurrent writers wait instead of raising immediately.
"""
def __init__(self, path: str, retention_days: int | None = None, busy_timeout_ms: int = 5000) -> None:
self.path = path
self.retention_days = retention_days
self.busy_timeout_ms = int(busy_timeout_ms)
# One connection per thread. A shared connection lets concurrent
# statements interleave, which was observed to let two callers both
# believe they were the first to accept the same event.
self._local = threading.local()
with self._connect_raw() as setup:
setup.executescript(_SCHEMA)
def _connect_raw(self) -> sqlite3.Connection:
try:
db = sqlite3.connect(self.path, isolation_level=None)
db.execute("PRAGMA journal_mode = WAL")
db.execute("PRAGMA synchronous = FULL")
db.execute(f"PRAGMA busy_timeout = {self.busy_timeout_ms}")
return db
except sqlite3.Error as exc:
raise BackendUnavailableError(f"cannot open audit store: {exc}") from exc
@property
def db(self) -> sqlite3.Connection:
conn = getattr(self._local, "conn", None)
if conn is None:
conn = self._local.conn = self._connect_raw()
return conn
@property
def retention_policy(self) -> RetentionPolicy:
return RetentionPolicy(
custody_class="development",
retention_days=self.retention_days,
immutable=False,
tamper_evidence=False,
durable=True,
)
def emit(self, event: AuditEvent) -> str:
"""Persist an event, generating no idempotency guarantee."""
return self.accept(event, payload_hash=_record_hash(event)).reference
def accept(self, event: AuditEvent, payload_hash: str) -> AcceptResult:
try:
validate_event(event)
except EventValidationError:
raise
reference = f"audit:{event.event_id}"
details = event.details if isinstance(event.details, dict) else {}
db = self.db
try:
# BEGIN IMMEDIATE takes the write lock up front, so the insert and
# the follow-up read are one atomic pair. Without it, two callers
# racing on the same event id can both be told they were first.
db.execute("BEGIN IMMEDIATE")
except sqlite3.Error as exc:
raise BackendUnavailableError(str(exc)) from exc
try:
inserted = db.execute(
"""
INSERT INTO events
(event_id, payload_hash, accepted_at, correlation_id, tenant, record)
VALUES (?, ?, ?, ?, ?, ?)
ON CONFLICT(event_id) DO NOTHING
RETURNING event_id
""",
(
event.event_id,
payload_hash,
datetime.now(timezone.utc).replace(microsecond=0).isoformat(),
str(details.get("correlation_id") or "") or None,
event.tenant,
json.dumps(event.as_record(), sort_keys=True),
),
).fetchone()
existing = None
if inserted is None:
existing = db.execute(
"SELECT payload_hash FROM events WHERE event_id = ?", (event.event_id,)
).fetchone()
db.execute("COMMIT")
except sqlite3.Error as exc:
_rollback(db)
raise BackendUnavailableError(str(exc)) from exc
except BaseException:
_rollback(db)
raise
if inserted is not None:
return AcceptResult(duplicate=False, reference=reference)
if existing is None:
# The row vanished between the insert and the read inside one
# transaction, which should be impossible. Retryable rather than
# guessed at.
raise BackendUnavailableError("event disappeared during accept")
if existing[0] != payload_hash:
raise EventConflictError(
f"event_id {event.event_id} already held with a different payload"
)
return AcceptResult(duplicate=True, reference=reference)
def health(self) -> None:
"""Raise :class:`BackendUnavailableError` if the store is unusable."""
try:
self.db.execute("SELECT 1 FROM events LIMIT 1").fetchone()
except sqlite3.Error as exc:
raise BackendUnavailableError(str(exc)) from exc
def close(self) -> None:
"""Close this thread's connection, if it has one."""
conn = getattr(self._local, "conn", None)
if conn is not None:
conn.close()
self._local.conn = None
def _rollback(db: sqlite3.Connection) -> None:
try:
db.execute("ROLLBACK")
except sqlite3.Error:
pass
def _record_hash(event: AuditEvent) -> str:
import hashlib
return hashlib.sha256(
json.dumps(event.as_record(), sort_keys=True).encode("utf-8")
).hexdigest()