audit-core/deploy/audit-core.yaml
tegwick 041a50aa78
All checks were successful
CI Smoke / host-smoke (push) Successful in 0s
CI Smoke / container-smoke (push) Successful in 1s
Pin factory-compatible audit receiver release
Assistant: codex
Assistant-Model: gpt-6-astra
Assistant-Session: 01a07ff8-19d0-7820-b4d0-1353833cb7fc
2026-09-11 06:48:47 +02:00

223 lines
9.1 KiB
YAML

# audit-core receiver — railiance01 (AUDIT-WP-0005-T03).
#
# Conventions match the deployed user-engine workload: digest-pinned image from
# forgejo.coulomb.social, non-root with a read-only root filesystem, and probes
# on a named http port.
#
# Apply order matters: the namespace label railiance.io/postgres-client is what
# platform-pg-consumer-ingress in the databases namespace admits, so without it
# the pod cannot reach the database at all.
---
apiVersion: v1
kind: Namespace
metadata:
name: audit-core
labels:
kubernetes.io/metadata.name: audit-core
railiance.io/workload-class: platform
# Admitted by NetworkPolicy platform-pg-consumer-ingress (rapp-postgres).
railiance.io/postgres-client: platform-pg
---
apiVersion: v1
kind: Service
metadata:
name: audit-core
namespace: audit-core
labels:
app.kubernetes.io/name: audit-core
spec:
type: ClusterIP
selector:
app.kubernetes.io/name: audit-core
ports:
- name: http
port: 8080
targetPort: http
protocol: TCP
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: audit-core
namespace: audit-core
labels:
app.kubernetes.io/name: audit-core
annotations:
# Rollback position. Update both together; `kubectl rollout undo` returns to
# the previous digest, and the schema note records whether that is safe.
audit-core.railiance.io/rollback-note: >-
Migrations 0001-0006 are additive. 0006 adds chain_hash/chain_prev and
then NOT NULL. An image that does not write those columns cannot accept
events after 0006. Do not roll back past sha256:7febc28e… to a pre-0007
writer. A future migration that drops or narrows a column must state
its own rollback position before it is released.
spec:
replicas: 1
revisionHistoryLimit: 5
strategy:
type: RollingUpdate
rollingUpdate:
maxUnavailable: 0
maxSurge: 1
selector:
matchLabels:
app.kubernetes.io/name: audit-core
template:
metadata:
labels:
app.kubernetes.io/name: audit-core
# Distinguishes receiver pods from the attestation CronJob's pods,
# which share the name label. NetworkPolicy audit-core-egress is
# scoped to this value so the receiver never gains the attest job's
# API-server reach (AUDIT-WP-0009-T02). Not in the Deployment's
# selector, which is immutable and does not need it.
app.kubernetes.io/component: receiver
spec:
securityContext:
runAsNonRoot: true
runAsUser: 10001
fsGroup: 10001
seccompProfile:
type: RuntimeDefault
# In-flight events must not be lost on rollout; the app shuts down
# gracefully on SIGTERM (AUDIT-WP-0004-T06).
terminationGracePeriodSeconds: 30
containers:
- name: audit-core
# REPLACE at release time with the built digest. A mutable tag is not
# an immutable image, and `:latest` must never be the only reference.
image: forgejo.coulomb.social/coulomb/audit-core@sha256:c82e0442de0fd181342916ae9cd5d6de41d859e1efda637bd93936c67873afa5
imagePullPolicy: IfNotPresent
ports:
- name: http
containerPort: 8080
env:
- name: AUDIT_CORE_HOST
value: "0.0.0.0"
- name: AUDIT_CORE_HTTP_PORT
value: "8080"
# Refuses to start on anything but operational (durable Postgres)
# custody. ``archive`` remains an accepted alias for one mixed
# rollout so an old manifest cannot refuse a new image.
- name: AUDIT_CORE_REQUIRE_CUSTODY_CLASS
value: operational
# Runtime role cannot CREATE TABLE. Schema changes are a Job
# with the migration lease (deploy/migrate-job.yaml).
- name: AUDIT_CORE_AUTO_MIGRATE
value: "0"
- name: AUDIT_CORE_DATABASE_SCHEMA
value: audit_core
- name: AUDIT_CORE_THREADS
value: "8"
- name: AUDIT_CORE_REQUEST_TIMEOUT
value: "30"
- name: AUDIT_CORE_DB_STATEMENT_TIMEOUT_MS
value: "30000"
# The database credential is a mounted directory, not a variable.
# A dynamic lease rotates while the pod runs; an env var is fixed at
# process start, so env delivery would force a restart — and a
# delivery gap — on every rotation. audit-core re-reads this
# directory on each connection attempt (AUDIT-WP-0005-T02).
- name: AUDIT_CORE_CREDENTIAL_DIR
value: /etc/audit-core/db
# The sender registry is read once at startup, so a variable is
# adequate here. Token rotation is overlap-first inside the
# registry itself and needs no restart either.
- name: AUDIT_CORE_SENDERS
valueFrom:
secretKeyRef:
name: audit-core-senders
key: senders.json
# Non-secret tenant/source scope. Tokens stay in the Secret;
# ExternalSecret refresh cannot shrink user-engine tenants
# below deploy/senders-scope.json.
- name: AUDIT_CORE_SENDERS_SCOPE_PATH
value: /etc/audit-core/senders-scope.json
# Chain-head attestation, written by the audit-core-attest CronJob
# and mounted read-only here (AUDIT-WP-0009-T02). The receiver
# reads it and never writes it: a receiver that could rewrite its
# own attestation could forge it, which is why the job is a
# separate workload with a separate identity.
#
# Absent, empty or stale degrades tamper_evidence rather than
# failing the pod — that is T01's intended behaviour and the
# correct state before the first run.
- name: AUDIT_CORE_ATTESTATION_PATH
value: /etc/audit-core/attestation/chain-head.json
resources:
requests:
cpu: 50m
memory: 64Mi
limits:
cpu: 500m
memory: 256Mi
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
# Custody lives in PostgreSQL. Nothing of value is written to the
# pod filesystem, and a read-only root enforces that rather than
# trusting it — the SQLite fallback cannot silently start
# accumulating audit records on ephemeral storage.
readOnlyRootFilesystem: true
volumeMounts:
- name: tmp
mountPath: /tmp
- name: database-credential
mountPath: /etc/audit-core/db
readOnly: true
- name: senders-scope
mountPath: /etc/audit-core/senders-scope.json
subPath: senders-scope.json
readOnly: true
# Mounted as a DIRECTORY, deliberately, and not with subPath.
# A subPath ConfigMap mount is resolved once at pod start and
# never updates — the daily attestation would land in the
# ConfigMap and never reach the running receiver, so
# tamper_evidence would age out to false while the job reported
# success every night. A directory mount is updated in place by
# the kubelet, and the backend re-reads the file per check.
- name: chain-head
mountPath: /etc/audit-core/attestation
readOnly: true
startupProbe:
httpGet: {path: /healthz, port: http}
periodSeconds: 3
failureThreshold: 20
readinessProbe:
# /readyz checks the backend is reachable and durable, so the pod
# leaves the Service when custody is unavailable rather than
# accepting events it cannot store.
httpGet: {path: /readyz, port: http}
periodSeconds: 10
timeoutSeconds: 2
failureThreshold: 3
livenessProbe:
# Deliberately /healthz, not /readyz: a database outage must not
# restart the pod in a loop. Losing readiness is the correct
# response; restarting solves nothing and loses in-flight work.
httpGet: {path: /healthz, port: http}
periodSeconds: 20
timeoutSeconds: 2
failureThreshold: 3
volumes:
- name: tmp
emptyDir: {}
- name: database-credential
secret:
# Kubernetes updates the projected files in place when the
# ExternalSecret refreshes, which is what makes restart-free
# rotation possible.
secretName: audit-core-database
# 0440 + fsGroup 10001: 0400 is root-only and the process cannot read it.
defaultMode: 0440
- name: senders-scope
configMap:
name: audit-core-senders-scope
defaultMode: 0444
- name: chain-head
configMap:
name: audit-core-chain-head
defaultMode: 0444
# optional: the pod must start before the first attestation exists.
optional: true