Add deployment manifests, custody-class guard and request counters
All checks were successful
CI Smoke / host-smoke (push) Successful in 0s
CI Smoke / container-smoke (push) Successful in 1s

AUDIT-WP-0005-T03 (progress). Manifests validated --dry-run=server
--validate=strict against railiance01; not applied, since deployment is gated
on RAPP-POSTGRES-WP-0002 and T02 credentials. Nothing here mutates the cluster.

Conventions read off the deployed user-engine workload rather than invented:
digest-pinned image from forgejo.coulomb.social, runAsNonRoot with
RuntimeDefault seccomp, no privilege escalation, all capabilities dropped,
readOnlyRootFilesystem, probes on a named http port, same resource envelope.

The namespace carries railiance.io/postgres-client: platform-pg, which is what
platform-pg-consumer-ingress in rapp-postgres admits; without that label the
pod cannot reach the database at all.

NetworkPolicies default-deny both directions, then permit ingress from the
user-engine namespace only, a separately labelled operator read path, and
egress to PostgreSQL in databases plus DNS.

Three decisions worth naming. Liveness is /healthz while readiness is /readyz,
so a database outage drops the pod from the Service rather than restarting it
in a loop. readOnlyRootFilesystem enforces the empty-filesystem property rather
than trusting it, so the SQLite fallback physically cannot accumulate audit
records on ephemeral storage. AUDIT_CORE_REQUIRE_CUSTODY_CLASS=archive makes a
missing database URL a startup failure instead of a silent downgrade to the
development store.

Counters deferred from WP-0004-T06 are exposed as JSON at /v1/stats behind the
read privilege, not as Prometheus exposition format: the cluster runs no
Prometheus, no ServiceMonitor CRD and no other scrape target, so an exposition
endpoint would target a scrape path that does not exist. Usable with curl now
and a small step from /metrics later.

Tests 77 -> 80.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
tegwick 2026-08-10 17:42:43 +02:00
parent 88d16847ff
commit 2f4e1adf66
6 changed files with 393 additions and 5 deletions

View file

@ -26,6 +26,7 @@ import logging
import os
import signal
import sys
import threading
from datetime import datetime, timezone
from http import HTTPStatus
from typing import Any
@ -52,6 +53,40 @@ MAX_BODY_BYTES = 256 * 1024
log = logging.getLogger("audit_core.ingestion")
class Counters:
"""In-process request counters (AUDIT-WP-0005-T03).
Exposed as JSON at ``/v1/stats`` rather than in Prometheus exposition
format, because railiance01 currently runs no Prometheus, ServiceMonitor
CRD, or any other scrape target. Building an exposition endpoint for a
scrape path that does not exist would be guessing; this is usable by an
operator with curl today and is a small step from a /metrics endpoint when
a metrics stack lands.
These reset on restart, which is correct for rate signals. The counters
that must survive a restart secret-shaped field findings are persisted
by the backend instead.
"""
FIELDS = ("accepted", "duplicate", "conflict", "rejected", "unauthorized",
"forbidden", "unavailable", "error")
def __init__(self) -> None:
self._lock = threading.Lock()
self._counts = {name: 0 for name in self.FIELDS}
self._started = datetime.now(timezone.utc).replace(microsecond=0)
def hit(self, name: str) -> None:
with self._lock:
if name in self._counts:
self._counts[name] += 1
def snapshot(self) -> dict[str, Any]:
with self._lock:
counts = dict(self._counts)
return {"since": self._started.isoformat(), "counts": counts}
class IngestionApplication:
"""WSGI application accepting user-engine outbox events.
@ -61,9 +96,20 @@ class IngestionApplication:
"""
def __init__(
self, backend: IdempotentAuditBackend, senders: SenderRegistry | str
self,
backend: IdempotentAuditBackend,
senders: SenderRegistry | str,
require_custody_class: str | None = None,
) -> None:
policy = backend.retention_policy
if require_custody_class and policy.custody_class != require_custody_class:
# Production sets this. Without it, losing AUDIT_CORE_DATABASE_URL
# silently downgrades custody to the development store instead of
# failing to start.
raise ValueError(
f"backend custody_class={policy.custody_class!r} does not meet the "
f"required {require_custody_class!r}; refusing to start"
)
if not policy.durable:
# The mock file backend declares durable=False. Refusing it here is
# what stops a development sink from silently becoming the
@ -78,6 +124,7 @@ class IngestionApplication:
senders = development_registry(senders)
self.backend = backend
self.senders = senders
self.counters = Counters()
def __call__(self, environ, start_response):
try:
@ -87,6 +134,7 @@ class IngestionApplication:
# start_response is never called and the sender sees a dropped
# connection it cannot classify.
log.exception("unhandled error in ingestion request")
self.counters.hit("error")
return self._json(
start_response, HTTPStatus.INTERNAL_SERVER_ERROR, {"error": "internal_error"}
)
@ -100,13 +148,14 @@ class IngestionApplication:
method = environ.get("REQUEST_METHOD")
identity = self.senders.authenticate(environ.get("HTTP_AUTHORIZATION"))
if identity is None:
self.counters.hit("unauthorized")
return self._json(
start_response, HTTPStatus.UNAUTHORIZED, {"error": "unauthorized"}
)
if method == "GET" and (
path.startswith("/v1/events")
or path in ("/v1/dead-letters", "/v1/secret-findings")
or path in ("/v1/dead-letters", "/v1/secret-findings", "/v1/stats")
):
return self._read(start_response, environ, path, identity)
@ -114,6 +163,7 @@ class IngestionApplication:
return self._json(start_response, HTTPStatus.NOT_FOUND, {"error": "not_found"})
if not identity.may_write:
self.counters.hit("forbidden")
return self._json(start_response, HTTPStatus.FORBIDDEN, {"error": "write_forbidden"})
raw = b""
@ -125,9 +175,11 @@ class IngestionApplication:
except SecretFieldRejection as exc:
self._count_secrets(payload, identity, "rejected", exc.findings)
self._dead_letter(raw, str(exc), identity)
self.counters.hit("rejected")
return self._json(start_response, HTTPStatus.BAD_REQUEST, {"error": str(exc)})
except (ValueError, TypeError, KeyError, json.JSONDecodeError) as exc:
self._dead_letter(raw, str(exc), identity)
self.counters.hit("rejected")
return self._json(start_response, HTTPStatus.BAD_REQUEST, {"error": str(exc)})
redaction = event.details.get("redaction")
@ -141,15 +193,18 @@ class IngestionApplication:
result = self.backend.accept(event, hashlib.sha256(raw).hexdigest())
except EventConflictError as exc:
log.warning("event conflict: %s", exc)
self.counters.hit("conflict")
return self._json(start_response, HTTPStatus.CONFLICT, {"error": "event_id_conflict"})
except EventValidationError as exc:
return self._json(start_response, HTTPStatus.BAD_REQUEST, {"error": str(exc)})
except BackendUnavailableError as exc:
log.error("backend unavailable: %s", exc)
self.counters.hit("unavailable")
return self._json(
start_response, HTTPStatus.SERVICE_UNAVAILABLE, {"error": "backend_unavailable"}
)
self.counters.hit("duplicate" if result.duplicate else "accepted")
return self._json(
start_response,
HTTPStatus.OK if result.duplicate else HTTPStatus.ACCEPTED,
@ -175,6 +230,8 @@ class IngestionApplication:
start_response, HTTPStatus.OK,
{"dead_letters": self.backend.dead_letters(_limit(query))},
)
if path == "/v1/stats":
return self._json(start_response, HTTPStatus.OK, self.counters.snapshot())
if path == "/v1/secret-findings":
return self._json(
start_response, HTTPStatus.OK,
@ -451,7 +508,11 @@ def main() -> None:
format='{"ts":"%(asctime)s","level":"%(levelname)s","logger":"%(name)s","msg":"%(message)s"}',
stream=sys.stdout,
)
app = IngestionApplication(build_backend(), SenderRegistry.from_env())
app = IngestionApplication(
build_backend(),
SenderRegistry.from_env(),
require_custody_class=os.environ.get("AUDIT_CORE_REQUIRE_CUSTODY_CLASS") or None,
)
serve(
app,
host=os.environ.get("AUDIT_CORE_HOST", "0.0.0.0"),