From e8035f3887fde35e26111a5b465dd05116ebaab6 Mon Sep 17 00:00:00 2001 From: tegwick Date: Tue, 18 Aug 2026 12:16:04 +0200 Subject: [PATCH] Build immutable policy publication artifact --- .forgejo/workflows/publish-image.yaml | 73 +++ .gitignore | 2 + .repo-classification.yaml | 20 + Containerfile | 30 ++ INTENT.md | 16 +- Makefile | 32 +- README.md | 49 +- build/index.html | 186 ++++++++ build/publication-manifest.json | 21 + .../standards/tenancy-posture/v0.1/index.html | 444 ++++++++++++++++++ .../v0.1/revisions/draft-8/index.html | 444 ++++++++++++++++++ build/tenancy-posture.html | 384 +-------------- deploy/nginx.conf | 38 ++ .../adr/ADR-0001-addressing-and-permanence.md | 60 +++ publication.json | 30 ++ tests/test_publication.py | 197 ++++++++ tests/test_release.py | 106 +++++ tools/build_site.py | 300 ++++++++++++ tools/check_currency.py | 38 ++ tools/render.py | 139 ++++-- tools/verify_release.py | 126 +++++ ...S-WP-0001-permanent-publication-surface.md | 92 +++- 22 files changed, 2375 insertions(+), 452 deletions(-) create mode 100644 .forgejo/workflows/publish-image.yaml create mode 100644 .gitignore create mode 100644 .repo-classification.yaml create mode 100644 Containerfile create mode 100644 build/index.html create mode 100644 build/publication-manifest.json create mode 100644 build/standards/tenancy-posture/v0.1/index.html create mode 100644 build/standards/tenancy-posture/v0.1/revisions/draft-8/index.html create mode 100644 deploy/nginx.conf create mode 100644 docs/adr/ADR-0001-addressing-and-permanence.md create mode 100644 publication.json create mode 100644 tests/test_publication.py create mode 100644 tests/test_release.py create mode 100644 tools/build_site.py create mode 100644 tools/check_currency.py create mode 100644 tools/verify_release.py diff --git a/.forgejo/workflows/publish-image.yaml b/.forgejo/workflows/publish-image.yaml new file mode 100644 index 0000000..bf42431 --- /dev/null +++ b/.forgejo/workflows/publish-image.yaml @@ -0,0 +1,73 @@ +# Uses the estate's tier-2 container-build runner and organization-scoped +# REGISTRY_USER / REGISTRY_TOKEN secrets. +name: Build and publish policy-nexus image + +on: + push: + branches: + - main + paths: + - ".forgejo/workflows/publish-image.yaml" + - "Containerfile" + - "deploy/**" + - "publication.json" + - "tests/**" + - "tools/**" + workflow_dispatch: + +env: + REGISTRY: forgejo.coulomb.social + IMAGE_NAME: coulomb/policy-nexus + DOCKER_HOST: tcp://127.0.0.1:2375 + +jobs: + build-and-push: + runs-on: container-build + steps: + - name: Build, verify, and publish immutable policy artifact + env: + REGISTRY_USER: ${{ secrets.REGISTRY_USER }} + REGISTRY_TOKEN: ${{ secrets.REGISTRY_TOKEN }} + run: | + set -eu + REF="${GITHUB_SHA:-main}" + SHORT="${REF:0:7}" + mkdir -p buildctx/_sources/net-kingdom "${HOME}/bin" + + wget -qO /tmp/policy-nexus.tar.gz \ + "https://forgejo.coulomb.social/${GITHUB_REPOSITORY}/archive/${SHORT}.tar.gz" + tar xzf /tmp/policy-nexus.tar.gz -C buildctx --strip-components=1 + + NETKINGDOM_REVISION=$(git ls-remote \ + https://forgejo.coulomb.social/coulomb/net-kingdom.git \ + refs/heads/main | awk '{print $1}') + test -n "$NETKINGDOM_REVISION" + NETKINGDOM_SHORT="$(printf '%s' "$NETKINGDOM_REVISION" | cut -c1-7)" + wget -qO /tmp/net-kingdom.tar.gz \ + "https://forgejo.coulomb.social/coulomb/net-kingdom/archive/${NETKINGDOM_SHORT}.tar.gz" + tar xzf /tmp/net-kingdom.tar.gz \ + -C buildctx/_sources/net-kingdom --strip-components=1 + + wget -qO- https://download.docker.com/linux/static/stable/x86_64/docker-27.3.1.tgz \ + | tar xz --strip-components=1 -C "${HOME}/bin" docker/docker + export PATH="${HOME}/bin:${PATH}" + docker version + echo "${REGISTRY_TOKEN}" | docker login "${REGISTRY}" \ + -u "${REGISTRY_USER}" --password-stdin + + IMAGE="${REGISTRY}/${IMAGE_NAME}" + docker build \ + --file buildctx/Containerfile \ + --build-arg "VCS_REVISION=${REF}" \ + --build-arg "NETKINGDOM_REVISION=${NETKINGDOM_REVISION}" \ + --tag "${IMAGE}:git-${REF}" \ + --tag "${IMAGE}:main" \ + buildctx + docker push "${IMAGE}:git-${REF}" + docker push "${IMAGE}:main" + + PUBLICATION_DIGEST=$(docker run --rm --entrypoint sha256sum \ + "${IMAGE}:git-${REF}" /usr/share/nginx/html/publication-manifest.json \ + | awk '{print $1}') + echo "published=${IMAGE}:git-${REF}" + echo "publication_manifest_digest=${PUBLICATION_DIGEST}" diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..43ae0e2 --- /dev/null +++ b/.gitignore @@ -0,0 +1,2 @@ +__pycache__/ +*.py[cod] diff --git a/.repo-classification.yaml b/.repo-classification.yaml new file mode 100644 index 0000000..46f8616 --- /dev/null +++ b/.repo-classification.yaml @@ -0,0 +1,20 @@ +repo_classification: + standard: Repo Classification Standard + version: "1.0" + classified_at: "2026-08-18" + classified_by: agent + category: project + domain: infotech + secondary_domains: + - government + capability_tags: + - canon + - governance + - knowledge + - documentation + business_stake: + - technology + - operations + business_mechanics: + - coordination + - operation diff --git a/Containerfile b/Containerfile new file mode 100644 index 0000000..1e62b0d --- /dev/null +++ b/Containerfile @@ -0,0 +1,30 @@ +FROM docker.io/nginxinc/nginx-unprivileged@sha256:65e3e85dbaed8ba248841d9d58a899b6197106c23cb0ff1a132b7bfe0547e4c0 AS runtime-base + +ARG VCS_REVISION=unknown +LABEL org.opencontainers.image.title="policy-nexus" \ + org.opencontainers.image.description="Canonical Coulomb policy publication surface" \ + org.opencontainers.image.source="https://forgejo.coulomb.social/coulomb/policy-nexus" \ + org.opencontainers.image.revision="$VCS_REVISION" + +COPY --chown=101:101 deploy/nginx.conf /etc/nginx/conf.d/default.conf + +USER 101:101 +EXPOSE 8080 + +FROM runtime-base AS local-artifact +COPY --chown=101:101 build/ /usr/share/nginx/html/ + +FROM docker.io/library/python@sha256:d09d15e60962ca365d1cd544a48773bac9d33f2fb1b00f2aa0deec78ade7dc31 AS release-builder +ARG NETKINGDOM_REVISION +ENV POLICY_NEXUS_SOURCE_REVISION_NET_KINGDOM=$NETKINGDOM_REVISION +WORKDIR /workspace/policy-nexus +COPY . /workspace/policy-nexus +COPY _sources/net-kingdom /workspace/net-kingdom +RUN rm -rf build \ + && python3 -m unittest discover -s tests -p 'test_*.py' \ + && python3 tools/build_site.py publication.json --output build \ + && python3 tools/verify_release.py build \ + && python3 tools/check_currency.py publication.json + +FROM runtime-base AS release-artifact +COPY --from=release-builder --chown=101:101 /workspace/policy-nexus/build/ /usr/share/nginx/html/ diff --git a/INTENT.md b/INTENT.md index da493a3..177e0c4 100644 --- a/INTENT.md +++ b/INTENT.md @@ -52,11 +52,11 @@ This repo exists so that policy has a permanent address and a known freshness. ## What it does not own -- **The content of estate policy.** Canon lives in `the-custodian`; per-repo - ADRs live in their repos. This repo publishes what those own and must never - become a second place where policy is edited. The local-files-are-source-of- - truth rule applies with full force: if the site and the source disagree, the - source is right and the publication is a defect. +- **The content of estate policy.** Canon lives in its owning canon repo; + per-repo ADRs live in their repos. This repo publishes what those own and + must never become a second place where policy is edited. The local-files-are- + source-of-truth rule applies with full force: if the site and the source + disagree, the source is right and the publication is a defect. - **Ratification.** Whether a draft becomes canon is a canon-process decision. This repo can show that a draft is in flight and how long it has been; it cannot advance it. @@ -94,9 +94,9 @@ publication surface with write authority becomes a second source of truth, and the estate has a standing rule against exactly that. The first content it must carry is already waiting: NetKingdom's *Tenancy -Posture* standard, which needs to reach six reviewing repos and is currently -served from a disposable artifact URL. The renderer that produces that page -from canon markdown now lives here as `tools/render.py`. +Posture* standard, now reviewed by all six affected repos and still served from +a disposable artifact URL. The renderer that produces that page from canon +markdown lives here as `tools/render.py`. ## What good looks like diff --git a/Makefile b/Makefile index 4271ed6..bf70359 100644 --- a/Makefile +++ b/Makefile @@ -1,15 +1,33 @@ -.PHONY: build clean +.PHONY: build check currency clean release-build release-check image-build # Publication targets. Source of truth is always the upstream repo; pages here # are generated and must never be hand-edited. -SRC_NETKINGDOM ?= $(HOME)/net-kingdom +SRC_NETKINGDOM ?= ../net-kingdom build: - mkdir -p build - python3 tools/render.py $(SRC_NETKINGDOM)/canon/standards/tenancy-posture_v0.1.md \ - --output build/tenancy-posture.html \ - --title "Tenancy Posture" \ - --subtitle "A framework for describing, holding and improving multi-tenancy — including where we are not there yet." + python3 tools/build_site.py publication.json --output build + +check: + python3 -m unittest discover -s tests -p 'test_*.py' + python3 -m py_compile tools/render.py tools/build_site.py tools/check_currency.py tools/verify_release.py + git diff --check + +currency: + python3 tools/check_currency.py publication.json clean: rm -rf build + +release-check: + python3 tools/verify_release.py build + python3 tools/check_currency.py publication.json + +release-build: clean build release-check + +image-build: release-check + @test -n "$(IMAGE_REF)" || (echo "IMAGE_REF is required" >&2; exit 2) + docker build \ + --file Containerfile \ + --target local-artifact \ + --build-arg VCS_REVISION=$$(git rev-parse HEAD) \ + --tag $(IMAGE_REF) . diff --git a/README.md b/README.md index 2504bda..0e67af7 100644 --- a/README.md +++ b/README.md @@ -1,21 +1,44 @@ # policy-nexus -Permanent publication and regulatory intake for the estate's policy surface. -Serves `policy.coulomb.social`. +Permanent publication for the estate's policy surface. Serves +`policy.coulomb.social`. -Two halves: +This repo publishes estate **canon and architecture decision records** from +the repositories that own them, at stable URLs, with visible status and +currency. Pages are generated, never authored here: the source of truth stays +upstream and this repo never writes back. -- **Outward** — publishes estate **canon and architecture decision records** - from the repositories that own them, at stable URLs, with visible status and - currency. Generated, never authored: the source of truth stays upstream and - this repo never writes back. -- **Inward** — records **regulation bearing on the estate**: rules that - constrain data it holds, markets it sells into, or obligations it has taken - on. A record says what a source said and when. It never says what the estate - must therefore do. +Regulatory intake and disclosure decisions belong to `risk-nexus`; publishable +records may arrive from it like any other source. This repo does not interpret +them. -Not a CMS, not a documentation site, not a civic-information corpus, and not a -source of legal advice. +Not a CMS, not a documentation site, not a policy author, and not a source of +legal advice. - Intent: `INTENT.md` - Workplans: `workplans/` + +Build and verify the publication locally with: + +```sh +make check +make build +make currency +``` + +`publication.json` is the explicit source and address registry. A build fails +closed when a source is unavailable or an immutable revision would change. + +Production publication is split from runtime ownership. This repository builds +and publishes the immutable OCI site image; `rapp-policy-nexus` owns the Helm +package, exposure checks, and rollback; `railiance-apps` selects the approved +production digests. A release build additionally refuses dirty or synthetic +source provenance: + +```sh +make release-build +make image-build IMAGE_REF=forgejo.coulomb.social/coulomb/policy-nexus:git-$(git rev-parse HEAD) +``` + +Tags are discovery handles only. Production always records the registry-resolved +OCI digest and the SHA-256 of `build/publication-manifest.json`. diff --git a/build/index.html b/build/index.html new file mode 100644 index 0000000..4c8bee8 --- /dev/null +++ b/build/index.html @@ -0,0 +1,186 @@ +Coulomb Policy Nexus
policy surfacegenerated from canonical sources — do not edit

Coulomb Policy Nexus

Canon and architecture decisions at stable addresses, with visible currency.

DocumentStatusLifecycleRevisionOwnerReviewedReview dueCurrency
NetKingdom Tenancy Posture v0.1proposedactivedraft-8net-kingdom2026-08-172027-02-17current
diff --git a/build/publication-manifest.json b/build/publication-manifest.json new file mode 100644 index 0000000..fd93e17 --- /dev/null +++ b/build/publication-manifest.json @@ -0,0 +1,21 @@ +{ + "documents": [ + { + "canonical_path": "standards/tenancy-posture/v0.1/index.html", + "currency": "current", + "id": "netkingdom-tenancy-posture", + "last_reviewed": "2026-08-17", + "lifecycle": "active", + "owner": "net-kingdom", + "review_due": "2027-02-17", + "revision": "draft-8", + "revision_path": "standards/tenancy-posture/v0.1/revisions/draft-8/index.html", + "source_digest": "99f802d91a0b3a65f0dac58230d8904f7c61cf3f81eff072fbbc59b634612a8a", + "source_revision": "cced59d3aa1dc0aa08fc128fc8c76699f59dcd90", + "status": "proposed", + "title": "NetKingdom Tenancy Posture v0.1" + } + ], + "generated_as_of": "2026-08-18", + "schema_version": 1 +} diff --git a/build/standards/tenancy-posture/v0.1/index.html b/build/standards/tenancy-posture/v0.1/index.html new file mode 100644 index 0000000..5c758cb --- /dev/null +++ b/build/standards/tenancy-posture/v0.1/index.html @@ -0,0 +1,444 @@ + + + + +NetKingdom Tenancy Posture v0.1 + +
netkingdom-tenancy-posture proposed · draft-8 net-kingdom reviewed 2026-08-17generated from canonical source — do not edit

NetKingdom Tenancy Posture v0.1

A framework for describing, holding and improving multi-tenancy — including where we are not there yet.

Source: net-kingdom · canon/standards/tenancy-posture_v0.1.md · cced59d3aa1dc0aa08fc128fc8c76699f59dcd90

Review due: 2027-02-17

Status

+

Proposed, draft-8; ratification-ready. Relocated from the-custodian/canon/architecture on 2026-08-17: multi-tenancy is part of the IT-security framework NetKingdom provides, so this framework belongs in NetKingdom canon beside the IAM Profile and the tenant-engine boundary contract, not in the work-factory canon.

+
  • draft-1 proposed a single model with fixed characteristics. Rejected: it could not describe a repo that is not there yet.
  • draft-2 reframed to graduated levels per axis. Externally corroborated (§16), but four of its statements were wrong and one thing it needed was missing.
  • draft-3 applied those corrections, added the retention axis, and recorded an adoption stance.
  • draft-4 closed the two gaps draft-3 left open: R4 had no mechanism beyond waiting, and the noisy-neighbour evidence artifact asserted something shared infrastructure cannot provide.
  • draft-5 relocated to NetKingdom and renamed the dimensions from planes to axes, because the word was already taken (§0).
  • draft-6 applied tenant-engine's review: five changes, including an axis that did not fit its data shape.
  • draft-7 applies audit-core, railiance-platform and flex-auth. Eleven further changes, two of them corrections to statements this document made as fact about other repos. Every posture I guessed was too generous, on every repo that has now self-reported.
  • draft-8 applies adaptive-pricing's review, the last of the six, and the consistency review across all declarations. It adds the missing availability axis, a canonical declaration schema, explicit authority for tier assurance, retention/placement coupling, downgrade propagation, and honest sanctioned customer language. It also corrects the distinction between an implemented control and an evidenced current level.
+

Reviewed by all six. The score: six repos found three live defects in their own code by reading the ladders — tenant-engine's unfiltered event accessor, audit-core's unfiltered read path, flex-auth's unauthenticated /v1/check — and railiance-platform found apps-pg running with no backup configured at all while writing its §10.2 disclosure. The framework changed to fit the repos; no repo was told to fabricate a posture.

+

Informed by five external research digests plus their index in the-custodian/research/2026-08-17-adr008-*, which carry full citations for the external claims made here.

+
+

00Terminology: axes, not planes

+

docs/platform-identity-security-architecture.md — accepted, 2026-07-23 — already uses plane for a trust and deployment layer: the bootstrap plane, the platform control plane, and tenant planes. That meaning is established, ratified, and owned by this repo.

+

Drafts 1–4 of this document, written elsewhere, used plane for something different: an independent dimension of concern. Two incompatible senses of one word inside one canon is exactly the concept-ownership collision the estate has been careful about elsewhere, and the newcomer yields.

+

This framework therefore describes six axes. They are orthogonal to NetKingdom's planes, not a subdivision of them:

+
  • A plane is where something runs and what trust it carries — bootstrap, platform control, tenant.
  • An axis is which property of tenancy is being described — identity, authorization, enforcement, placement, retention, availability.
+

A workload in the tenant plane has a position on all six axes. A platform control plane service does too. The two vocabularies compose and neither replaces the other.

+

The rename is also an improvement. A posture vector is literally a point in six-dimensional space, and "axis" says that where "plane" did not.

+
+

01Context

+

Drafts 1–4 opened by claiming the estate "has never written down what it is building". Relocation proved that wrong, and the correction is worth keeping visible: docs/platform-identity-security-architecture.md has described the trust model, the tenant model and a capability progression since 2026-07-23. The accurate claim is narrower — what was missing is a way to say how far a given service has got, and to hold several answers at once. Seven documents cover slices of the subject and none of them does that:

+
DocumentCoversStatus
iam-profile_v0.3 (NetKingdom)Tenant identifier shape, tenant_roles claim, staleness rulesRatified
tenant-engine-boundary-contract_v0.1 (NetKingdom)Who owns tenant records, roles, plan assignmentRatified
business-app-service-contract_v0.1 §1 (Custodian)Business apps: instance-per-client, tenant-keyed dataRatified
rapp-postgres ADR-0001Consumer + tenant isolation in PostgreSQLProposed, governs one repo
rapp-postgres ADR-0002Per-consumer retention and the erasure horizonProposed, governs one repo
shared-platform-relational-storage_v0.1The stacked-boundary gapRouted 2026-08-10, still unratified
platform-identity-security-architecture (NetKingdom)Trust model, planes, tenant model, capability progressionAccepted 2026-07-23
+

This document is downstream of that architecture and must not restate it. It answers one question the architecture leaves open: given the model, where is this particular service today, and how would anyone know?

+

Four failures existed when drafting began.

+

The gap was diagnosed once and the fix stalled. The v0.1 draft was written to fill this hole and has sat unratified in neither canon directory. §20 attaches a ratification path so this one does not join it.

+

Placement was owned by nobody. user-engine-pg and target-revenue-pg are dedicated; apps-pg, net-kingdom-pg, platform-pg, state-hub-db and forgejo-db are shared. Both live, neither written down. tenant-engine raised this with railiance-platform on 2026-08-16. Draft-8 resolves the authority split in §8.2.

+

Two contradictory defaults were already ratified. Business apps get instance-per-client; platform services pool. Nothing says which shape a new service takes, and no definition separates the categories. Decision 4.4.1 now supplies the default; §19.4 retains the missing classification rule.

+

There is no honest way to describe a repo that is not there yet. The estate absorbs repos with weak or absent tenant separation. Today such a repo is simply non-conformant, leaving it two bad options: misrepresent its posture, or stay outside the framework.

+
+

02What this document is

+

A framework, not a model. It specifies no single correct implementation. It supplies terminology (§3, §4), a declaration (§5), a conformance rule (§6), methodology (§12), and evidence definitions (§13).

+

A service is conformant when its declared posture is accurate and its trajectory recorded. A service is non-conformant when it claims a level it cannot evidence — regardless of how high or low that level is.

+
+

03Six orthogonal axes

+

"Is this multi-tenant?" is treated as one question. It is six, and they are independent:

+
AxisQuestionVocabulary owner
Identity (I)How is a tenant named and validated?tenant-engine / IAM Profile
Authorization (A)How is a request bound to the tenants it may act for?flex-auth
Enforcement (E)Where, mechanically, is the tenant boundary enforced?This framework
Placement (P)Which substrate holds a tenant's data?railiance-platform
Retention (R)How long does data persist, and how is it erased?The storage platform; policy by the consumer
Availability (V)What failure can the complete service path survive, and within what recovery objective?The delivering service; substrate facts by its providers
+

Conflation produces errors today. rapp-postgres's PostgresConsumer carries tenantIsolation: consumer-service-boundary — an E-axis fact in a P-axis artifact, reading as though storage enforces something it does not. The "dedicated versus shared" argument mixes P (capacity, blast radius) with E (correctness).

+

The axes are separated precisely so each may sit at a different level.

+

Decision 3.1: every document, declaration and plan tier that says "isolation" MUST name which axis it means.

+

Decision 3.2: the axes couple at their tops and the couplings MUST be stated where they apply, not used to argue the axes are one:

+
  • E4 is reachable only at P3 or above.
  • R's erasure horizon is bounded below by P — on shared substrate, a consumer's horizon is the instance maximum (§4.5).
  • R4 by key destruction is bounded by the key boundary, which is an E-axis property. Shredding a single tenant's data requires the application to encrypt under a per-tenant key before writing; the storage platform cannot supply it. Reaching the top of the retention ladder is not a retention project.
  • V composes as the minimum across the critical request path, not the maximum of its components. A replicated application on a single-instance database is not V2. A tested degraded mode may remove a dependency from that path, but the bypass itself is part of the V evidence.
+

Decision 3.3 — scope. The P and R ladders describe a service's primary datastore. The V ladder describes the service's complete critical request path, including providers it synchronously depends on. Caches, search indices, message queues and background jobs are named leak surfaces in the external baselines and are assessed separately, not silently covered by a datastore level. A declaration names material secondary stores and asynchronous paths as exceptions rather than implying that one vector proves them safe.

+
+

04Graduated levels

+

Each axis carries an ordered ladder. Higher is stronger, not better: the right level is the one a service can evidence and its risk warrants.

+

4.1 Identity (I)

+
+
Identityaxis I
I0No tenant concept. Data not attributable to a tenant.
I1A local tenant notion exists but is not canonical, or the tenant is taken from the request rather than from a verified token.
I2Canonical identifiers, bound at the identity provider and carried as a verified claim, and verified by this service on its own inbound calls.
I3I2 plus capability roles honoured, with live tenant-engine re-query for privileged, destructive, credential-vending or aal2-class decisions.
Ladder ends at I3.
+
+

I1 now explicitly absorbs request-supplied tenant identifiers. "Never trust client-supplied tenant IDs without validation" is a named anti-pattern; a service reading the tenant from a header is at I1 however canonical the string.

+

An axis is assessed on a service's own inbound surface, never on its authority over the concept. tenant-engine is the source of existence for tenant records and is nonetheless at I1, because it takes the acting identity from the request body rather than from a verified token. Draft-5 conflated these by naming the authority inside the I2 definition, which made the level describing canonical identity unclaimable by the service that provides it. Corrected on tenant-engine's review — a reader would otherwise assume the authority must be at I2 by definition.

+

business-app-service-contract §2.1 sets app-local accounts as the v1 baseline for business apps — a sanctioned low level with recorded triggers for moving up. That is the pattern this framework generalises.

+

4.2 Authorization (A)

+
+
Authorizationaxis A
A0No authorization, or tenant context not carried.
A1Ad-hoc checks scattered through handlers.
A2A single local authorization boundary; tenant context bound once, centrally.
A3Decisions delegated to flex-auth as PDP, with live re-query where the IAM Profile requires it.
A4A3 over a standard PDP interface (OpenID AuthZEN Authorization API 1.0), so the decision point is swappable and the enforcement point is not coupled to one engine's request shape.
+
+

This ladder describes enforcement points. A decision point cannot occupy A3 — "delegated to flex-auth" is not something flex-auth can do. A service that is a PDP declares two numbers: its own inbound level, and the maximum it enables for consumers. flex-auth reads A0, enables A3 — accurate, and considerably more alarming than A3, which is the point. Raised by flex-auth, whose absence from the §5 worked examples was this surfacing implicitly.

+

A4 is new. The specification reached Final in January 2026 and Keycloak shipped experimental support in May; the argument for it is interoperability — a swappable decision point and an enforcement point not coupled to one engine's request shape.

+

Correction from flex-auth's review: earlier drafts also justified A4 as ending the copying of action strings between repos. It does not. AuthZEN standardises the envelope — subject, action, resource, context, endpoint — and deliberately does not standardise the action vocabulary or the policy language. At A4, tenant.guardrail.set still has to be agreed and still gets copied. Those are two problems with different fixes, and the cheaper one is not A4: flex-auth's registry already carries action definitions per system and could serve them read-only. The vocabulary argument is withdrawn.

+

Internal service-to-service calls are in scope for this axis. "Skipping tenant validation for internal services" is a named anti-pattern and our estate is mostly internal calls.

+

Correction from flex-auth's review: earlier drafts asserted that flex-auth calls tenant-engine synchronously on the authorization path. That is not true. The adapter is built and complete and has no non-test caller, so the IAM Profile's live re-query exists and is unwired — which is also why flex-auth cannot reach I3. Built-and-unwired is the worst of the three states because it reads as capability.

+

The requirement, narrowed on their proposal because the original was too strong to be met and would have made tenant-engine a hard availability dependency of every decision in the estate:

+

Tenant context MUST be carried on every internal hop and MUST NOT be re-derived from a service identity. It MUST be revalidated against tenant-engine at least once per request chain — at the service that holds or mutates the tenant's data, or before a privileged, destructive, credential-vending or aal2-class decision, whichever comes first. A hop that neither holds tenant data nor makes such a decision may carry the context without revalidating it.

+

And carrying tenant context is worthless without an authenticated hop to carry it over. flex-auth found this in itself: it carries tenant context faithfully and cannot distinguish "user-engine asking on behalf of tenant X" from "any pod asking on behalf of tenant X".

+

4.3 Enforcement (E)

+
+
Enforcementaxis E
E0None. Data not tenant-keyed; separation incidental or absent.
E1Data tenant-keyed, filtering applied per query at call sites.
E2Filtering centralised at a single service-side choke point binding authenticated identity to permitted tenants.
E3E2 plus platform-assisted filtering: row-level security keyed on a tenant GUC set transaction-locally, or an equivalent enforced data-access layer.
E4Structural: the credential a workload holds cannot address another tenant's data at all. Requires per-tenant credentials and per-tenant substrate.
+
+

Correction from draft-2. Draft-2 described E3 as something "the application cannot trivially route around". That is false and it was this document overclaiming in exactly the way §6 prohibits. Any session can re-issue SET on a custom GUC, so an attacker with SQL execution can reset the tenant and read across the boundary. What E3 buys is precise, and the ladder must say so:

+
ThreatE1E2E3E4
A developer forgets a tenant predicate
A new code path bypasses the choke point
SQL injection reaching the connection
The application process is compromised
+

E3 is a strong control against accident — the common case, and the one that causes real breaches — and no control at all against compromise. Only E4 holds against both, because the credential itself cannot address another tenant's data.

+

Correction: E3 layers on E2, it does not replace it. External practice treats application-layer and database-layer filtering as complementary. A service that dropped its choke point on reaching E3 would be worse off, since E3 fails open under injection. Claiming E3 therefore requires the E2 evidence artifact as well.

+

Correction: the GUC is set transaction-locally. Draft-2 said "at pool checkout", which is session scope and the wrong instrument. Under a pooler in statement mode, SET leaks between clients and returns other tenants' rows — a failure that appears only under production concurrency and produces no error. Use SET LOCAL inside an explicit transaction.

+

Platform enforcement is a platform obligation. Reaching E3 requires the storage platform to offer the mechanism: provisioned policies, a documented GUC contract, and a probe. Where a consumer wants E3 and the platform has not supplied it, the gap is the platform's. §19.6 asks rapp-postgres to define that contract, which must carry FORCE ROW LEVEL SECURITY on every tenant table (without it the table owner bypasses policies silently, and ADR-0001 already established that our migration role owns the tables it creates), no BYPASSRLS on leased roles, SECURITY INVOKER for ordinary logic, and an EXPLAIN comparison because RLS disables functional indexes built on non-leakproof functions.

+

Not all data is tenant-keyed, and the ladder must not pretend otherwise. A registry whose rows are the tenants has no per-tenant predicate to scope a policy by; enforcing one would break the service's function rather than secure it. tenant-engine's tenants table is the worked example — key-cape enumerates it at token issuance and flex-auth queries it live, both of which are cross-tenant reads by design.

+

A service with mixed data shapes declares E-level plus a registry exception: the level its tenant-keyed tables hold, and a named list of tables excluded because they are registries rather than tenant data. The exception is part of the claim and is reviewable; an unnamed exception is an overclaim. Without this, mixed-shape services either overclaim or stay at E2 permanently, and tenant-engine declined to claim E3 on precisely that reasoning.

+

Default expectation for a new platform service: E2 at first serve, E3 recorded as target. Services whose cross-tenant exposure would be a reportable breach SHOULD target E3 or above.

+

4.4 Placement (P)

+
+
Placementaxis P
P0Shares a database with another consumer.
P1Database per consumer, shared cluster.
P2Dedicated cluster per consumer.
P3Dedicated cluster per tenant.
P4P3 plus separate region or jurisdiction.
+
+

Enforcement and placement are independent axes. Plotted together, with where each service actually sits — parenthesised entries are targets or defaults rather than current positions, and marks a cell the coupling in §3.2 makes unreachable:

+
Enforcement →
E4
business app
E3
target
E2
tenant-engine
audit-core
E1
absorbed repo
E0
P0
P1
P2
P3
P4
Where a service sits todayTarget or defaultUnreachable at this placement
+

P0 → P1 → P2 is movement along the horizontal axis only. Those steps buy consumer isolation, capacity predictability, independent retention and a smaller operational blast radius. They do not raise the tenant boundary by one step. Only P3 makes E4 reachable. This is the most misusable fact in the framework and §11 governs how it may be described.

+

Decision 4.4.1: P1 is the default for platform services; P3 for client-facing business apps, as already ratified. A service unsure which it is must resolve that first (§19.4).

+

Decision 4.4.2 — placement scopes to data substrate. Identity-provider placement (realm-per-tenant versus Organizations) is the same silo/pool decision on a different substrate, is live in our estate, and is undecided. Realm-per-tenant carries a stated ceiling around 5–20 tenants, far below our target. Recorded here as a parallel question (§19.7), not folded into P.

+

4.5 Retention and erasure (R)

+

New in draft-3. Implemented abstractly by the storage platform for any dataset; policy is built on top of that interface by the consumer or its governance layer. Reference implementation: rapp-postgres ADR-0002.

+
+
Retentionaxis R
R0No retention or deletion position. Data kept indefinitely by default; no deletion path exists.
R1Platform default retention applies (N=30 days). The consumer has declared no requirement.
R2Retention declared as N days per dataset; the erasure horizon is published, and the consumer makes no promise shorter than it.
R3Policy-driven deletion: the consumer or its governance layer declares what is due, the platform sweeps whole datasets on that instruction and evidences each run.
R4Verified erasure: data proven unrecoverable across live storage, backups and derived copies, by one of the two routes below.
+
+

R4 has two routes and a service MUST name which one it uses.

+
RouteMechanismCost
Horizon-elapsedWait out the published erasure horizon; the data ages out of every retained copy.Available to everyone, proves little, and the wait is set by a co-resident's retention requirement rather than your own.
Key-destroyedEncrypt per entity, then destroy the key. Retained copies survive but are unreadable.Requires per-entity keys, strong encryption, and an auditable destruction record. Immediate.
+

Decision 4.5.3 — key destruction is not sufficient on its own. The key-destroyed route requires that no retained commitment reveals the erased content. Found by audit-core, and it is a general defect rather than a fact about them:

+
  • A SHA-256 over a canonical record whose fields are low-entropy — event type, actor, tenant, subject, timestamp — is a confirmation oracle. Anyone holding the hash can guess the payload, hash the guess, and confirm a match. Destroying the key does not make the content unrecoverable while that hash survives.
  • Shreddability is not retrofittable onto an integrity chain that commits to cleartext. It has to be built as encrypt-then-hash at accept time, with the chain committing to ciphertext. Retrofitting means rewriting the chain — the exact thing a tamper-evident log exists to make detectable.
+

So a service claiming R4 by key destruction must show that its retained commitments — hashes, chains, indexes, search keys — do not reveal what was erased. The remedies are an HMAC under a per-subject key that dies with the key, or a per-record salt destroyed alongside it. audit-core cannot reach R4 under its current design and targets R2; a fleet R4 target must exempt it explicitly.

+

Regulatory standing of the key-destroyed route, stated carefully because overclaiming here is worse than anywhere else in this framework. Data protection authorities have accepted key destruction as erasure where physical deletion would be manifestly disproportionate, and the practice is recognised under conditions — strong encryption, irreversible destruction, and an auditable record of it. The EDPB has not formally endorsed it as Article 17 erasure. A service reaching R4 by key destruction is making a defensible claim, not a settled one, and must say so rather than reporting a clean "deleted".

+

Three further properties.

+

The erasure horizon is the interval between deleting data and it ceasing to be recoverable from anything the platform holds. Deleting a row does not remove it from yesterday's backup. With an N-day window, deleted data remains recoverable for N days. That is the difference between "deleted" and "erased" and the estate had never written it down.

+

On shared substrate, retention is not per-consumer. Physical backup is instance-wide — one WAL stream, one window — so the instance retention is derived as the maximum across co-resident consumers, and every consumer's horizon is that maximum. A consumer declaring 7 days beside one declaring 90 gets 90. This is the retention analogue of ADR-0001's blast-radius disclosure: state the coupling rather than imply an isolation that is not there.

+

Retention is therefore a placement trigger. A consumer needing a horizon shorter than the instance floor cannot have one at P1. It moves to P2 for a reason with nothing to do with performance — which is exactly why it needs recording, since nobody looks for a retention argument when reviewing placement.

+

Decision 4.5.4 — a retention promise binds both R and P. A tier making a retention claim records an R minimum and a maximum erasure horizon in days. It also requires P2 or above unless its provider contract guarantees that the shared-substrate horizon stays within that maximum and rejects or notifies before a co-resident change would extend it. A bare R2 minimum is insufficient: at P1 another consumer can change the promise without changing the tier or its holder.

+

Deletion splits mechanism from policy. The platform deletes whole datasets on instruction and records an opaque policy reference it never interprets, so every deletion traces to what authorised it. Rows are not a dataset: row expiry is the consumer's own DML under its migration lease. Dropping a consumer's whole database is an operator-gated offboarding step, never a scheduled one.

+

4.6 Availability (V)

+

New in draft-8. adaptive-pricing found that §11 required availability claims to map to a minimum level while the framework supplied no availability vocabulary. Placement is not a substitute: a dedicated cluster can still be a single instance on a single node.

+
+
Availabilityaxis V
V0No availability or recovery position. Recovery is untested or depends on improvisation.
V1Restart or recreate recovery in one failure domain is documented and exercised. Interruption is expected; this is recovery, not failover.
V2Redundant instances provide automated service failover, with measured RTO/RPO; a shared failure domain or critical dependency may remain.
V3The complete critical path survives loss of one declared failure domain, with measured RTO/RPO from an exercise.
V4The complete critical path survives regional loss through tested multi-region failover, with measured RTO/RPO.
+
+

Decision 4.6.1 — V is end-to-end. A service declares the minimum across the components and synchronous providers required to serve the operation. An application with three replicas over a V1 database is V1. A status page or replica count is not evidence of a higher level.

+

Decision 4.6.2 — availability claims name the operation. A read-only degraded mode and a mutation path may have different V levels. Decision 5.2 applies: declare the paths and quote the minimum unless the customer-facing claim explicitly and unambiguously names the narrower operation.

+
+

05The posture vector

+

A service states one level per axis, plus a target, review dates, evidence and any exceptions. current is the highest evidenced level; a control present in code but still awaiting the evidence required by §13 goes in implemented, not in current:

+
schema_version: "0.1"
+framework: netkingdom-tenancy-posture
+service: example-service
+role: tenant-data-service
+tenancy:
+  current:     { I: 2, A: 3, E: 2, P: 1, R: 1, V: 1 }
+  implemented: { E: 3 }
+  target:      { I: 2, A: 3, E: 3, P: 1, R: 2, V: 2 }
+  reviewed: "2026-08-17"
+  review_due: "2027-02-17"
+  service_class: interactive
+  gap:
+    E: "RLS is implemented; the §13 E3 probe is still absent."
+    R: "Retention declared; erasure horizon not yet published to consumers."
+    V: "Automated failover is not implemented or exercised."
+evidence:
+  A3: "docs/evidence/authorization-denial.md"
+  E2: "docs/evidence/cross-tenant-review.md"
+  P1: "rapp-postgres/docs/evidence/isolation-2026-08-10.md"
+

Placement exceptions. Draft-2 assigned one P level per service, which cannot express the vertically partitioned model — most tenants pooled, some dedicated — that §11's isolation tiers require. A tier requiring P2 bought by three tenants would put the service at two levels at once, forcing an over- or under-claim. Placement is therefore declared as a default plus exceptions:

+
  placement_exceptions:
+    - tenants: ["tenant:enterprise:*"]
+      P: 3
+      reason: "isolation tier; see adaptive-pricing tier definition"
+

A service with exceptions must be able to say which tenants are on which substrate. That mapping is a first-class artifact, not archaeology.

+

Decision 5.5 — a provider declares what it makes reachable, not where it sits. The six ladders describe a consumer of infrastructure. They describe a provider of it badly, and railiance-platform's review demonstrated how badly: apps-pg is I0 A0 E0 because a database has no tenant concept, carries no tenant claim and applies no tenant predicate. Those zeros are structural, not weak — the cluster is exactly as strong as its consumers make it.

+

The sharp case is OpenBao at E0. Literally correct, and actively misleading: the mechanism in place is credential-scoped structural separation — E4 machinery — pointed at a consumer boundary rather than a tenant one. A reader scanning a column of E values would rank it below a service doing per-query filtering in application code, inverting the real security position.

+

So a platform service additionally declares, per axis, the level available now, the maximum it can make reachable, and what a consumer must do to reach it. For apps-pg: E4 unreachable (shared credential per consumer, no per-tenant credential), E3 conditional on the GUC contract, R2 blocked on a backup target, V1 at most on the single-node rail. That is the sentence a consumer actually needs, and no arrangement of the consumer ladders produces it.

+

A provider's own P is n/a, not a number. apps-pg provides P1; it is not at P1, and writing P: 1 there would later read as an isolation claim.

+

Worked examples after applying the evidence rule and minimum-across-paths rule consistently:

+
ServiceCurrentNotes
tenant-engineI1 A0 E1 P n/a R0 V0Acting identity is caller-supplied; unauthorised read paths set the A minimum; E2-shaped child-table controls are not evidenced; SQLite is outside P; no erasure or availability evidence. This corrects draft-7, which quoted A2/E2 despite its own minimum/evidence rules.
audit-coreI1 A2 E1 P1 R2 V0E2 is implemented on both paths but awaits the adversarial artifact, so current remains E1. Its 30-day retention and erasure horizon are now declared and published.
flex-authI1 A0 E1 P n/a R n/a V0Enables A3 for consumers. /v1/check authenticates no caller; E2 is implemented but not evidenced.
platform-pg (provider)I0 A0 E0 P n/a R2 V1Provides P1; backup/restore and single-node recovery are evidenced. Provides no tenant boundary by itself.
apps-pg (provider)I0 A0 E0 P n/a R0 V0Zeros are structural, except R0/V0 are live gaps: no backup and no recovery evidence.
adaptive-pricing observatoryI0 A0 E0 P n/a R n/a V0Local, unauthenticated, single-user analysis surface; not a production service.
A newly absorbed repoI1 A1 E1 P0 R0 V0Conformant if declared, with a recorded path.
+

Decision 5.1: the posture vector is declared in the repo, not in the hub, consistent with local-files-are-source-of-truth.

+

Decision 5.2 — declare per path, quote the minimum. A service whose mutations are authorized and whose reads are not is at the reads' level. The quoted number is the minimum across paths; the per-path detail is declared beside it.

+

Draft-6 required only the minimum, on tenant-engine's review. audit-core then showed why that is insufficient on its own: a bare minimum destroys signal, because E3-write/E1-read declares identically to E1/E1. Bare per-path invites "our write path is E3", which is the sentence §6 exists to stop. Both, related explicitly, is the rule.

+

Two services found this shape in themselves within a day of each other — tenant-engine (writes authorized, three read routes not) and audit-core (write path tenant-filtered, read path not filtered at all). Most services enforce harder on write than read, so this is the common case, not the corner.

+

Decision 5.3 — n/a is a level, and it is conformant. P0 presupposes a shared database and R0 presupposes retained data. A service holding nothing at rest — flex-auth runs with its registry and policy baked read-only into the image and no decision log persisted — is neither. A datastore outside a ladder's substrate vocabulary, such as tenant-engine's current SQLite PVC, also uses n/a rather than inventing a level. Without an admissible n/a, a missing rung forces the fabrication §6 prohibits, which is precisely what draft-1 was rejected for. n/a is declared with a stated reason.

+

Decision 5.4 — the vector lives at tenancy.yaml in the repo root. Draft-6 said "in the repo" and not where or in what shape, which left §12's guard needing per-repo archaeology. flex-auth adopted tenancy.yaml speculatively; adopted here as the convention. A repo representing one service uses the single-service form above. A layer repo uses the schema's services list in the same root file — one vector per service, never an average. The normative schema is canon/schemas/tenancy-posture_v0.1.schema.json; prose documents may explain a declaration but do not replace it. The schema carries current, implemented, target, reviewed, review_due, gap, placement_exceptions, service_class (§8.3), per-path detail (§5.2), and provider reachability (§5.5). From the net-kingdom repo, owners validate one or more declarations with uv run tools/tenancy-posture/validate.py <path>...; the validator applies the JSON Schema and the evidence, date, implemented/current and provider-range rules that JSON Schema alone cannot express.

+
+

06Conformance is accuracy, not altitude

+

A service is conformant when its declared posture is accurate, its target is recorded, and it does not claim a level it cannot evidence. It is non-conformant when it overclaims — at any altitude.

+
  • Declaring E0 is conformant. Concealing E0 is not.
  • A repo may be absorbed at any posture. It may not be absorbed silently.
  • No service is blocked from the estate for being low on a ladder. Services MAY be blocked from specific work — serving a tenant grouping, holding a data class, carrying a plan tier — by requirements expressed as minimum levels.
  • Downgrading is permitted and must be declared. A regression found by guarding is a defect; a regression declared in advance is a decision.
  • A low level may be permanent by design, and the declaration must be able to say so. flex-auth is I1 and always will be: a decision point evaluates the claims it is handed, and verifying its own inputs would make it the identity provider its scope refuses to be. A target equal to current with a reason is a settled position, not a stalled trajectory, and §12's guard must not nag it as though it were one.
+

Decision 6.1 — downgrades propagate. Before a planned downgrade of a current level or a provider's available level, the declaring repo MUST resolve the tier definitions and consumers that reference it. A downgrade below a recorded minimum blocks the change until the claim is changed, the workload is moved, or the affected owner explicitly accepts the gap. An unplanned regression is an incident and triggers the same notifications. Updating tenancy.yaml without notifying dependants is declaration drift, not a completed downgrade.

+

Without the axis separation, "not rigorous about tenant separation" is one verdict a repo passes or fails. With it, the same repo is I1 A1 E1 P0 R0 V0 with a path — a plan, not an indictment.

+
+

07Portability across placement levels

+

Movement between P levels must be operational, not a rebuild:

+
  • Connect by injected credential only — no cluster, host, namespace or database name in source.
  • Own a whole database, never tables inside someone else's.
  • Idempotent schema creation.
  • No cross-database joins or co-location assumptions.
+

Decision 7.1: mandatory at P1 and above. At P3, SHOULD rather than MUST — a per-client instance that never moves is not misconformant for naming its own database.

+
+

08Placement triggers

+

Recorded at provisioning time: noisy neighbour on a latency-critical path; a compliance or residency requirement; a plan tier requiring a higher minimum; an erasure horizon that no longer fits (§4.5); connection or memory ceiling reached.

+

Decision 8.1: triggers MUST be monitored, not merely recorded. A trigger in a YAML comment nobody re-reads is documentation, not control.

+

Decision 8.2 — split authority, machine-reconciled. railiance-platform owns the placement rule; the package repo owns the substrate numbers and enforcement; the consuming repo owns its workload requirements; adaptive-pricing owns any tier minimum. adaptive-pricing declined a standing co-signature and the framework accepts the replacement: typed tier minima are joined to consumer and provider declarations at tier definition and whenever one changes. A machine-checkable constraint must not depend on somebody remembering to collect a signature.

+

Decision 8.2.1 — trigger monitoring has an owner. The provider monitors capacity ceilings and co-residency; the consumer monitors latency, compliance and erasure requirements; adaptive-pricing monitors tier-definition changes. The placement owner reconciles those signals. A trigger marked unmonitored is an explicit gap and cannot support a customer assurance claim.

+

8.3 Service class — a placement input, never a priority

+

A latency-critical consumer and a batch consumer can share an instance today with nothing distinguishing them. tenant-engine sits on flex-auth's synchronous authorization path and chose a 5s statement timeout for that reason; audit-core, co-resident, is not latency-critical. Nothing prioritises between them.

+

The framework does not add a QoS axis, because the platform cannot enforce one. Community PostgreSQL has no resource governor: no per-role CPU or I/O priority, no resource queues, no workload classes. Those exist in EDB's enterprise variant, in Greenplum, and in SQL Server — not in what we run. A declared priority level would therefore be an unenforced claim sitting in a declaration, which is precisely what retiring tenantIsolation was about. An axis implies graduation and enforcement; this has neither.

+

Decision 8.3.1 — co-residents are equal. On shared substrate no consumer's query yields to another's. A consumer whose latency requirement cannot survive an unprioritised neighbour must escalate to P2. That is the honest mechanism and it is the only one we have.

+

Decision 8.3.2 — service class is declared anyway, as a category rather than a level: latency-critical, interactive, or batch. It buys three things, none of which is priority:

+
  • A placement input. Mixing latency-critical with batch on one instance is a recognised mismatch. It may still be the right call — it is right today — but it should be a decision, not an accident of who was provisioned when.
  • A trigger. A latency-critical consumer acquiring a batch co-resident is a recorded placement trigger under §8, on the same footing as noisy neighbour.
  • An acceptance criterion for evidence. The noisy-neighbour artifact in §13 asks whether measured degradation is acceptable; without a declared class that word has no referent. Degradation tolerable for batch may be an outage for latency-critical.
+

Decision 8.3.3 — class mixture must be visible. The platform reports which classes are co-resident. An unenforceable risk that nobody can see is strictly worse than one that is stated.

+

The known escalation short of P2 is gateway-level prioritisation — ordering submissions in a connection proxy by the requesting tenant's current consumption. It is real, it is where the industry puts this when it must, and it is new infrastructure we do not run. Recorded as the option, not adopted.

+
+

09Credentials as a tenancy control

+

Short-lived leased credentials re-read at connection checkout, with overlap-first rotation, bound the residual risk at every E level below E4: a leaked credential expires rather than persisting. Stronger than the industry norm of a long-lived per-service secret.

+

Decision 9.1: static long-lived database credentials are not a sanctioned path for any service above E0.

+

Decision 9.2 — the rule extends to consumer-facing credentials. Draft-6 named database access only. audit-core pointed out that its ingest credentials are static long-lived bearer tokens, rotated by publishing a second alongside the first — and that the argument applies with more force to the credential that actually carries the tenant claim than to the one that reaches the database behind it. Read as an accidental omission; it was. Consumer-facing credentials are named in. Where a service cannot yet meet this, it is a stated gap rather than a silent exclusion.

+
+

10Blast radius must be published

+

Decision 10.1: every platform holding consumer data MUST publish, in concrete terms, what a leaked runtime credential can and cannot reach at the levels it operates. rapp-postgres ADR-0001 §5 is the reference. Where the model cannot provide a guarantee, the platform says so and names the escalation.

+

Decision 10.2 — quotas are disclosed, not discovered. The same obligation extends from what a leaked credential can reach to what the platform will refuse to do for you. Every consumer MUST be told, at provisioning, the throttles and quotas enforced against it — connection limits, statement timeouts, idle-transaction timeouts — and told again when they change. A consumer learning its statement timeout by hitting it in production is a disclosure failure, not a consumer bug. This is how tenant-engine was provisioned, by good practice rather than by rule; the rule now exists.

+
+

11Commercial expression

+
  • 11.1 Plan tiers are expressed internally as typed assurance requirements. A tier may require E3 P2 R2 V2 and a maximum erasure horizon; it need not print those labels anywhere customer-facing.
  • 11.2 Marketing and product language is free. No requirement to expose level labels or this document. "Dedicated infrastructure", "isolated tenancy", "private instance" all remain available.
  • 11.3 The constraint is on evidence, not vocabulary. A customer-facing isolation, availability or retention claim must map to a minimum level the delivering service actually holds, recorded once when the tier is defined. The review is internal and happens at tier definition — not per campaign.
  • 11.4 Two hard lines, because these reach contracts and compliance questionnaires:
  • A claim that another tenant cannot reach the customer's data requires E4.
  • A claim that deleted data is gone requires R4, or an erasure horizon disclosed alongside it. Where R4 is reached by key destruction, the claim is defensible but not settled law (§4.5) — it may be made, and it may not be made in language that implies a regulator has blessed it.
  • A claim that service survives loss of a zone requires V3; regional-loss language requires V4. "High availability" without a named failure and measured recovery objective is not an assurance claim this framework can evidence.
  • 11.5 — sanctioned honest language. The strong prohibitions above must not leave a commercial writer with only silence:
  • E3 may be described as database-backed defence against an omitted tenant filter; it must not be paraphrased as "another tenant cannot reach".
  • P2 may be described as a dedicated service database cluster with an independent capacity and restore boundary; it is not tenant-dedicated.
  • R2 may state the declared retention and published erasure horizon.
  • V1 may state exercised restart recovery in one failure domain and must say that interruption and single-domain loss remain.
  • 11.6 — authority and reconciliation. The tier definition is authoritative for the minimum and customer wording. tenancy.yaml is authoritative for the delivering service's current level; provider declarations are authoritative for what infrastructure makes available. None is derived by copying another. Approval joins them and fails closed on a missing, stale or insufficient declaration. A performance-differentiated tier requires P2 or an enforceable resource governor; service class alone grants no priority.
+
+

12Methodology — analyze, establish, improve, guard

+

Analyze. Assess a repo against the ladders; produce tenancy.current with reasoning recorded. Applies to new and absorbed services alike.

+

Establish. Declare the target and gap. The target is set by data class, tenant groupings served and plan tiers carried — not by ambition.

+

Improve. Move one axis at a time. Raising P while leaving E untouched is the characteristic misstep.

+

Guard. Verify continuously that the declared posture holds — against the service's own declaration, not a universal maximum. Nobody must prove every service is at E4; the check is that none is below what it declared.

+

Regression found by guarding is a defect; regression declared in advance is a decision. The estate has been bitten twice by silent pin rollbacks producing ordinary-looking 403s and 404s rather than errors. Posture regression looks the same — an RLS context leak returns correct-looking rows for the wrong tenant. Guarding must be designed for invisible failure, not for crashes.

+
+

13Evidence per level

+

Decision 13.1: a current level is claimed only with its evidence artifact present. This turns §6's accuracy rule from an honour system into a check. implemented records a control observed in code or configuration whose required artifact is still absent; it never satisfies a tier minimum.

+

Decision 13.1a — the floor needs no artifact, only a reason. Found independently by audit-core and flex-auth: the table below defines artifacts from I2, A2, E1, P1, R2 upward and none below, so a literal 13.1 made the lowest rungs unclaimable — including §5's own worked example of a conformant absorbed repo, I1 A1 E1 P0 R0, which could not satisfy it on any axis. A rule that forbids the declaration §6 exists to permit is a defect in the rule.

+

At I0/I1, A0/A1, E0, P0, R0/R1, V0, or n/a, a declaration requires a stated reason, not an artifact. Evidence is what stops you overclaiming, and there is nothing to overclaim at those floors.

+

Decision 13.4 — an artifact must assert something achievable. Draft-3's noisy-neighbour evidence required proof that a saturating consumer "does not breach" another's allowance. Shared infrastructure cannot provide that; the risk is inherent and cannot be wholly removed. An artifact that can only fail, or that passes by being run gently enough, is an overclaim wearing the costume of evidence. Where a property cannot be guaranteed, the artifact measures and records it instead.

+

Decision 13.2 — evidence is of two kinds, and conflating them is an overclaim. Mechanical evidence is a structural assertion a machine can make and belongs in CI. Adversarial evidence is semantic, requires setting up separate tenant contexts and comparing responses, and carries a review date rather than a green build. Cross-tenant findings are the category external testing practice identifies as needing human review. A passing CI run is not E2 evidence.

+
LevelEvidenceKind
I2Identifiers validated against the vocabulary; rejection test for a malformed id; binding shown to come from a verified tokenMechanical
I3Live re-query demonstrated on an aal2-class path; cached-claim path shown unused thereMechanical
A2Choke point identified; test that an unbound request is refusedMechanical
A3Live decision with a denial observed at the endpoint, not only at the decision surfaceMechanical
A4Decision served over the standard interface; a second PDP substituted without PEP change, with the decision differences between the two recorded — substitution proves interface portability, not decision equivalenceMechanical
E1Every tenant-owned table carries the tenant keyMechanical
E2Choke point identified; identity bound to tenant A demonstrably cannot read tenant BAdversarial, with a review date
E3FORCE ROW LEVEL SECURITY on every tenant table; no BYPASSRLS on leased roles; probe that a session without the GUC reads nothing; probe that a wrong GUC reads nothing; EXPLAIN comparisonMechanical
E4Per-tenant credential demonstrated unable to connect to another tenant's substrateMechanical
P1–P4Provisioning declaration plus the platform's isolation probesMechanical
Shared P1–P2 capacity assuranceA recorded baseline of per-consumer resource usage; a run in which one consumer saturates its declared allowance; evidence that the governance controls bind (the greedy consumer is held at its limits) and that the degradation co-residents experience is measured, recorded and judged acceptable against each one's declared service class (§8.3); the aggregate headroom at time of measurementAdversarial, load-generated, with a review date
R2Declared retention rendered; erasure horizon published and reported in the operator surfaceMechanical
R3Sweep evidence records: timestamp, dataset, identifiers removed, authorising policy referenceMechanical
R4Erasure demonstrated across live data, backups and derived copies within the horizonAdversarial
V1Critical dependencies enumerated; restart/recreate recovery exercised; interruption and measured recovery time recordedMechanical exercise
V2One instance terminated while traffic continues or recovers automatically; measured RTO/RPO and remaining shared failure domains recordedAdversarial, failure-injected
V3Declared failure domain removed in an exercise; complete critical path and degraded modes observed against RTO/RPOAdversarial, failure-injected
V4Region made unavailable in an exercise; traffic and state recover in the alternate region against RTO/RPOAdversarial, failure-injected
+

The P1–P4 artifact proves the declared placement topology. The shared-capacity artifact is additional: it is required before a P1/P2 service can claim that a noisy-neighbour control binds, that the trigger is actively guarded, or that a customer performance assurance survives co-residency. It is not required merely to report the true topology as P1 or P2. No such capacity artifact exists in the estate today, so §11 requires P2 or an enforceable governor for a performance-differentiated tier.

+

Decision 13.3: the tenant-boundary E2/E3 and noisy-neighbour artifacts do not exist anywhere in the estate today. rapp-postgres runs 19 adversarial probes, all against the consumer boundary, none against the tenant boundary inside a consumer. Externally, what this framework calls a tenant boundary failure is Broken Object Level Authorization — OWASP API1, top of the API Security Top 10 since that list launched, and the most commonly exploited API vulnerability in published assessments. We have no coverage for the highest-ranked risk in our class of system. §19.3 records the owner.

+
+

14Adoption stance — structure, not tooling

+

Decision 14.1: external research is design input. This estate adopts published standards and structural patterns; it does not adopt tooling unless that tooling is an established industry standard with broad application. Everything else is built ground-up, so it can be optimised and refactored as the estate sees fit.

+
ClassStance
Security baselines (OWASP Multi-Tenant Security Cheat Sheet, API Security Top 10)Adopt as the external reference our ladders answer to
Standards bodies (OpenID AuthZEN 1.0)Adopt — this is what A4 is
Reference taxonomies (Azure tenancy models, AWS SaaS Lens, cell architecture)Adopt as structure
Engine behaviour (PostgreSQL RLS mechanics)Facts, not tooling
Third-party analyzers and test frameworksDo not adopt. Take their rule taxonomies as checklists for probes we write ourselves
+

The practical effect is small and good: rapp-postgres already owns a ground-up probe harness — bash and psql, no dependency tree — that found four real defects in its own provisioning SQL. The evidence artifacts in §13 become new probes in a tool we control. One idea worth reimplementing from the external survey is policy-diff classification: labelling a change to an enforcement policy as safe or breaking before it lands.

+
+

15Alternatives considered

+

One fixed model with a single set of characteristics (draft-1). Rejected: cannot describe a repo that is not there yet, forcing absorbed repos to misrepresent their posture or stay outside. A framework that can only describe its own end state is not a framework.

+

A maturity model with a single overall level. Rejected: collapses the axis separation. A service strong on identity and weak on enforcement has a specific, actionable gap; one composite score hides it and invites averaging.

+

Prohibiting row-level security (draft-2's inherited position). Rejected in draft-2, refined in draft-3: RLS is a real rung against the common threat. The error was never RLS — it was describing E3 in E4's language.

+

Schema-per-consumer in one database. Rejected: pg_catalog is readable per-database, so every co-resident enumerates every other's table and column names regardless of grants. Retained as a describable state, never a target.

+

Mandating E4 for everyone. Rejected: the tenant taxonomy includes consumer (private individuals) and family. A cluster per private individual is economically impossible; the taxonomy is itself evidence pooling is required.

+

Per-consumer physical backup retention. Rejected: CNPG retention is a property of the instance's WAL archive. There is no mechanism, and claiming it would be a fabricated guarantee. Hence the derived maximum in §4.5.

+

Platform-scheduled row expiry. Rejected: requires the platform to hold DML authority over consumer schemas and interpret consumer data semantics, both forbidden by ADR-0001. The consumer's migration lease is the correct instrument.

+

Leaving each repo to its own model. Rejected: the status quo, which produced two contradictory ratified defaults and an unowned placement question.

+
+

16Held against outside practice

+

The graduated reframe is corroborated, not invented here. Microsoft's tenancy-model guidance states it almost verbatim: "Instead of viewing isolation as a discrete property, consider it a spectrum. You can deploy components of your architecture that are more isolated or less isolated than other components in the same architecture." The same guidance derives our E↔P coupling independently — shared deployment means enforcement lives in application code; dedicated deployment means it is structural.

+

Stronger than typical. Most multi-tenancy literature models one boundary, tenant-to-tenant. This estate has two stacked boundaries: platform-service to platform-service, and tenant to tenant inside a consumer. Naming them separately and refusing to enforce both with one mechanism is uncommon and correct. Graduated per-axis levels also beat the silo/pool/bridge trichotomy, which is approximately our P axis with the other four missing — which is why it cannot express "pooled infrastructure, structurally enforced boundary".

+

Weaker than typical. The pool model's standard mitigation is a verified enforcement layer every service is demonstrably routed through. We have the concept and none of the verification (§13.3).

+

Adopted without naming it. Short-lived leased credentials re-read at checkout beat the long-lived-secret norm. §9 promotes it to a tenancy control.

+

Still unexplored. Neither P nor R describes a cell — a slice of infrastructure with a fixed maximum size, sized so one cell's failure is survivable and cell count scales linearly. platform-pg is, in these terms, an uncapped cell: §17 computes a ceiling and nothing enforces it (§19.8).

+

Sources: the five research digests in the-custodian/research/2026-08-17-adr008-*, which carry full citations for the claims in this section.

+
+

17Scaling demands

+

Derived from the live platform-pg specification. Connection arithmetic is exact; the per-backend memory estimate remains unmeasured and is explicitly a gap in rapp-postgres ADR-0004.

+
instances:        1              (no HA; single-node rail)
+max_connections:  100
+memory limit:     1Gi
+per consumer:     14 connections (12 runtime + 2 migration)
+

The hard connection bound is roughly six declarations; the enforceable operational ceiling is four. Seven declarations request 98 of 100 connections before CNPG's instance manager, metrics exporter and reserved slots. Every one of them is politely inside its declared 14-connection allowance; the instance still fails. ADR-0004 sets four because memory is expected to bind first and fails by OOM-killing every co-resident rather than refusing one connection.

+

That distinction matters because our governance addresses the wrong shape. Per-consumer connection_limit, statement_timeout and idle_in_transaction_session_timeout guard well against one greedy consumer. They do nothing about the aggregate of many modest ones, which is the second and less intuitive noisy-neighbour failure and the one this number describes. Two workload consumers plus the isolation probe occupy three of the four declared slots. The next workload request must trigger measurement and the overflow decision before admission.

+

Memory likely binds first. 100 backends against 1Gi is ~10MB per backend. Connection exhaustion errors clearly; memory pressure OOM-kills and degrades every co-resident at once.

+

E3 and pooling. Corrected from draft-2, which had this backwards. Transaction-scoped context (SET LOCAL inside an explicit transaction) is what makes E3 safe under a pooler. Statement-level pooling is what breaks it, serving other tenants' rows under concurrency with no error. E3 constrains which pooling mode is available, not whether pooling is available.

+

Retention consumes the volume. WAL accumulates with the window, and §4.5 makes the window the maximum across consumers. A consumer declaring a long retention extends everyone's horizon and everyone's storage draw against a 20Gi volume.

+

Restore time couples all consumers. Physical backup is instance-wide, so a consumer's RTO is a function of total instance size, not its own.

+

platform-pg is V1. instances: 1 on a single-node rail provides exercised restart recovery and no failover. P1 describes its consumer placement and says nothing about this availability fact; the new V axis carries it.

+
+

18Consequences

+
  • The estate gains one vocabulary and a way to be honest about partial adoption.
  • Absorbed repos get a described state and a path instead of a failing grade.
  • tenantIsolation in PostgresConsumer is revealed as a mislabelled field.
  • The verification problem becomes tractable: guard against declaration.
  • Draft-2's RLS prohibition is reversed and its E3 description corrected; rapp-postgres acquires an obligation to define and offer the mechanism.
  • Adding a consumer with long retention silently extends everyone's erasure horizon. This must reach the consumer review checklist, not only this document.
  • A service selling an isolation tier must maintain a tenant→substrate mapping it does not have today.
  • Availability becomes an end-to-end, evidenced property rather than an inference from replica count or placement.
  • Nothing here changes a running system.
+
+

19Review resolutions and residual questions

+
  1. tenantIsolation field — resolved. rapp-postgres retired it. A consumer declaration asks for mechanisms; posture lives in the consumer's tenancy.yaml.
  2. Placement ownership — resolved. §8.2 records the split. The policy has one owner; typed tier requirements replace the declined commercial co-signature.
  3. E2, E3 and noisy-neighbour evidenceowned as of 2026-08-17 by whitehat-security (WHITEHAT-WP-0001), an independent adversarial evidence facility seeded for this purpose. audit-core and tenant-engine were right to decline it as fleet-scope work; the answer was a home of its own rather than a volunteer.
+

Owned by NetKingdom — corrected 2026-08-17; an earlier revision of this section proposed otherwise on independence grounds and was overruled. Offensive security is security work and belongs with the repo that owns security. The facility is framed offensively rather than as a conformance checker: it is pointed at infrastructure we choose, our own estate among them, and conformance testing is one use of a general capability.

+

The residual tension is recorded rather than resolved: NetKingdom owns this framework and the facility that tests conformance to it, so those findings are NetKingdom assessing NetKingdom. The mitigation is that findings leave for risk-nexus, under the-custodian, rather than being closed in place. Proportionate, not perfect. Revisit if conformance findings start getting quietly closed.

+

Two consequences land back here. Cadence is now a security parameter, not a schedule — for any control whose guarantee is detection rather than prevention, the interval between probe runs is the exposure window, and rapp-postgres ADR-0003 leaves that number to the facility. And a passing suite is not proof of isolation; it is proof that the attacks attempted did not work. §13's evidence artifacts should be read with that distinction, because a green run recorded as "E2 verified" would be exactly the overclaim §6 prohibits.

+
  1. Business app vs platform service — open. Custodian canon: a classification rule. Candidate: reuse repo-classification-standard_v1.0.
  2. Tier → minimum level mapping — policy resolved, implementation open. adaptive-pricing owns typed minima and wording; tenant-engine owns plan assignment by id. Current tiers make no assurance claims.
  3. The E3 mechanism — resolved. rapp-postgres ADR-0003 publishes the GUC contract with the FORCE/BYPASSRLS/SECURITY INVOKER/EXPLAIN requirements.
  4. Identity-provider placement — open. Owner of key-cape: realm-per-tenant or Organizations? Realm-per-tenant's ~5–20 tenant ceiling is below our target.
  5. Cell sizing — resolved for platform-pg. rapp-postgres ADR-0004 sets four consumers and names absent overflow target platform-pg-2; measurement and provisioning remain live gaps.
  6. Retention floor and ceiling — resolved as policy. Both exist; requests outside them fail validation and the package repo owns the numbers.
  7. Engine neutrality — open. The P ladder rests on a PostgreSQL property. State it engine-specifically and say so, or abstract it and risk a non-Postgres implementation that silently differs?
  8. Erasure versus audit — framework resolved. audit-core: crypto-shredding a tenant's audit records destroys the evidence the service exists to hold, and ADR-0001 §2 deliberately built the role model so history could not be rewritten. The usual resolution separates the fact of an event, retained, from its personal payload, encrypted per subject and shreddable. Raised because a naive "R4 everywhere" target would instruct the audit service to destroy its own evidence. audit-core targets R2 and is explicitly not a fleet R4 target. The legal basis for retaining audit facts remains a risk/legal question outside this framework.
  9. Quality of serviceresolved 2026-08-17. Co-residents are equal; a declared service class informs placement but never grants priority. See §8.3. The question asked whether to add a QoS dimension; the answer is no, and the reason is that we could not enforce one.
+

Routed elsewhere, deliberately. The tenant identifier tenant:<grouping>:<name> embeds headcount bands (small, medium, large) that change as a tenant grows, contradicting the consensus that identifiers should not encode mutable attributes. That is a critique of ADR-0013, not of this framework, and belongs to tenant-engine and NetKingdom canon. Folding it in here would overreach.

+
+

20Ratification path

+
  1. Reviewed by tenant-engine, flex-auth, audit-core, rapp-postgres, railiance-platform and adaptive-pricing against §19. Complete in draft-8.
  2. Each publishes its own posture vector (§5) as part of review. The framework is validated by whether it can describe them accurately — if a repo cannot express itself in these six ladders, the ladders are wrong and this document changes, not the repo. Complete in draft-8; all six root declarations validate against the canonical schema.
  3. On acceptance, supersedes the routing of rapp-postgres/docs/canon-drafts/shared-platform-relational-storage_v0.1-draft.md, whose §§3–8 are absorbed here. That draft is withdrawn rather than left pending.
  4. On acceptance, rapp-postgres ADR-0001 through ADR-0004 move to accepted and are annotated as the PostgreSQL implementation of the E, P, R and shared-capacity rules.
+
netkingdom-tenancy-posture · draft-8 · proposednet-kingdom · canon/standards/tenancy-posture_v0.1.md · cced59d3aa1dc0aa08fc128fc8c76699f59dcd90
diff --git a/build/standards/tenancy-posture/v0.1/revisions/draft-8/index.html b/build/standards/tenancy-posture/v0.1/revisions/draft-8/index.html new file mode 100644 index 0000000..5c758cb --- /dev/null +++ b/build/standards/tenancy-posture/v0.1/revisions/draft-8/index.html @@ -0,0 +1,444 @@ + + + + +NetKingdom Tenancy Posture v0.1 + +
netkingdom-tenancy-posture proposed · draft-8 net-kingdom reviewed 2026-08-17generated from canonical source — do not edit

NetKingdom Tenancy Posture v0.1

A framework for describing, holding and improving multi-tenancy — including where we are not there yet.

Source: net-kingdom · canon/standards/tenancy-posture_v0.1.md · cced59d3aa1dc0aa08fc128fc8c76699f59dcd90

Review due: 2027-02-17

Status

+

Proposed, draft-8; ratification-ready. Relocated from the-custodian/canon/architecture on 2026-08-17: multi-tenancy is part of the IT-security framework NetKingdom provides, so this framework belongs in NetKingdom canon beside the IAM Profile and the tenant-engine boundary contract, not in the work-factory canon.

+
  • draft-1 proposed a single model with fixed characteristics. Rejected: it could not describe a repo that is not there yet.
  • draft-2 reframed to graduated levels per axis. Externally corroborated (§16), but four of its statements were wrong and one thing it needed was missing.
  • draft-3 applied those corrections, added the retention axis, and recorded an adoption stance.
  • draft-4 closed the two gaps draft-3 left open: R4 had no mechanism beyond waiting, and the noisy-neighbour evidence artifact asserted something shared infrastructure cannot provide.
  • draft-5 relocated to NetKingdom and renamed the dimensions from planes to axes, because the word was already taken (§0).
  • draft-6 applied tenant-engine's review: five changes, including an axis that did not fit its data shape.
  • draft-7 applies audit-core, railiance-platform and flex-auth. Eleven further changes, two of them corrections to statements this document made as fact about other repos. Every posture I guessed was too generous, on every repo that has now self-reported.
  • draft-8 applies adaptive-pricing's review, the last of the six, and the consistency review across all declarations. It adds the missing availability axis, a canonical declaration schema, explicit authority for tier assurance, retention/placement coupling, downgrade propagation, and honest sanctioned customer language. It also corrects the distinction between an implemented control and an evidenced current level.
+

Reviewed by all six. The score: six repos found three live defects in their own code by reading the ladders — tenant-engine's unfiltered event accessor, audit-core's unfiltered read path, flex-auth's unauthenticated /v1/check — and railiance-platform found apps-pg running with no backup configured at all while writing its §10.2 disclosure. The framework changed to fit the repos; no repo was told to fabricate a posture.

+

Informed by five external research digests plus their index in the-custodian/research/2026-08-17-adr008-*, which carry full citations for the external claims made here.

+
+

00Terminology: axes, not planes

+

docs/platform-identity-security-architecture.md — accepted, 2026-07-23 — already uses plane for a trust and deployment layer: the bootstrap plane, the platform control plane, and tenant planes. That meaning is established, ratified, and owned by this repo.

+

Drafts 1–4 of this document, written elsewhere, used plane for something different: an independent dimension of concern. Two incompatible senses of one word inside one canon is exactly the concept-ownership collision the estate has been careful about elsewhere, and the newcomer yields.

+

This framework therefore describes six axes. They are orthogonal to NetKingdom's planes, not a subdivision of them:

+
  • A plane is where something runs and what trust it carries — bootstrap, platform control, tenant.
  • An axis is which property of tenancy is being described — identity, authorization, enforcement, placement, retention, availability.
+

A workload in the tenant plane has a position on all six axes. A platform control plane service does too. The two vocabularies compose and neither replaces the other.

+

The rename is also an improvement. A posture vector is literally a point in six-dimensional space, and "axis" says that where "plane" did not.

+
+

01Context

+

Drafts 1–4 opened by claiming the estate "has never written down what it is building". Relocation proved that wrong, and the correction is worth keeping visible: docs/platform-identity-security-architecture.md has described the trust model, the tenant model and a capability progression since 2026-07-23. The accurate claim is narrower — what was missing is a way to say how far a given service has got, and to hold several answers at once. Seven documents cover slices of the subject and none of them does that:

+
DocumentCoversStatus
iam-profile_v0.3 (NetKingdom)Tenant identifier shape, tenant_roles claim, staleness rulesRatified
tenant-engine-boundary-contract_v0.1 (NetKingdom)Who owns tenant records, roles, plan assignmentRatified
business-app-service-contract_v0.1 §1 (Custodian)Business apps: instance-per-client, tenant-keyed dataRatified
rapp-postgres ADR-0001Consumer + tenant isolation in PostgreSQLProposed, governs one repo
rapp-postgres ADR-0002Per-consumer retention and the erasure horizonProposed, governs one repo
shared-platform-relational-storage_v0.1The stacked-boundary gapRouted 2026-08-10, still unratified
platform-identity-security-architecture (NetKingdom)Trust model, planes, tenant model, capability progressionAccepted 2026-07-23
+

This document is downstream of that architecture and must not restate it. It answers one question the architecture leaves open: given the model, where is this particular service today, and how would anyone know?

+

Four failures existed when drafting began.

+

The gap was diagnosed once and the fix stalled. The v0.1 draft was written to fill this hole and has sat unratified in neither canon directory. §20 attaches a ratification path so this one does not join it.

+

Placement was owned by nobody. user-engine-pg and target-revenue-pg are dedicated; apps-pg, net-kingdom-pg, platform-pg, state-hub-db and forgejo-db are shared. Both live, neither written down. tenant-engine raised this with railiance-platform on 2026-08-16. Draft-8 resolves the authority split in §8.2.

+

Two contradictory defaults were already ratified. Business apps get instance-per-client; platform services pool. Nothing says which shape a new service takes, and no definition separates the categories. Decision 4.4.1 now supplies the default; §19.4 retains the missing classification rule.

+

There is no honest way to describe a repo that is not there yet. The estate absorbs repos with weak or absent tenant separation. Today such a repo is simply non-conformant, leaving it two bad options: misrepresent its posture, or stay outside the framework.

+
+

02What this document is

+

A framework, not a model. It specifies no single correct implementation. It supplies terminology (§3, §4), a declaration (§5), a conformance rule (§6), methodology (§12), and evidence definitions (§13).

+

A service is conformant when its declared posture is accurate and its trajectory recorded. A service is non-conformant when it claims a level it cannot evidence — regardless of how high or low that level is.

+
+

03Six orthogonal axes

+

"Is this multi-tenant?" is treated as one question. It is six, and they are independent:

+
AxisQuestionVocabulary owner
Identity (I)How is a tenant named and validated?tenant-engine / IAM Profile
Authorization (A)How is a request bound to the tenants it may act for?flex-auth
Enforcement (E)Where, mechanically, is the tenant boundary enforced?This framework
Placement (P)Which substrate holds a tenant's data?railiance-platform
Retention (R)How long does data persist, and how is it erased?The storage platform; policy by the consumer
Availability (V)What failure can the complete service path survive, and within what recovery objective?The delivering service; substrate facts by its providers
+

Conflation produces errors today. rapp-postgres's PostgresConsumer carries tenantIsolation: consumer-service-boundary — an E-axis fact in a P-axis artifact, reading as though storage enforces something it does not. The "dedicated versus shared" argument mixes P (capacity, blast radius) with E (correctness).

+

The axes are separated precisely so each may sit at a different level.

+

Decision 3.1: every document, declaration and plan tier that says "isolation" MUST name which axis it means.

+

Decision 3.2: the axes couple at their tops and the couplings MUST be stated where they apply, not used to argue the axes are one:

+
  • E4 is reachable only at P3 or above.
  • R's erasure horizon is bounded below by P — on shared substrate, a consumer's horizon is the instance maximum (§4.5).
  • R4 by key destruction is bounded by the key boundary, which is an E-axis property. Shredding a single tenant's data requires the application to encrypt under a per-tenant key before writing; the storage platform cannot supply it. Reaching the top of the retention ladder is not a retention project.
  • V composes as the minimum across the critical request path, not the maximum of its components. A replicated application on a single-instance database is not V2. A tested degraded mode may remove a dependency from that path, but the bypass itself is part of the V evidence.
+

Decision 3.3 — scope. The P and R ladders describe a service's primary datastore. The V ladder describes the service's complete critical request path, including providers it synchronously depends on. Caches, search indices, message queues and background jobs are named leak surfaces in the external baselines and are assessed separately, not silently covered by a datastore level. A declaration names material secondary stores and asynchronous paths as exceptions rather than implying that one vector proves them safe.

+
+

04Graduated levels

+

Each axis carries an ordered ladder. Higher is stronger, not better: the right level is the one a service can evidence and its risk warrants.

+

4.1 Identity (I)

+
+
Identityaxis I
I0No tenant concept. Data not attributable to a tenant.
I1A local tenant notion exists but is not canonical, or the tenant is taken from the request rather than from a verified token.
I2Canonical identifiers, bound at the identity provider and carried as a verified claim, and verified by this service on its own inbound calls.
I3I2 plus capability roles honoured, with live tenant-engine re-query for privileged, destructive, credential-vending or aal2-class decisions.
Ladder ends at I3.
+
+

I1 now explicitly absorbs request-supplied tenant identifiers. "Never trust client-supplied tenant IDs without validation" is a named anti-pattern; a service reading the tenant from a header is at I1 however canonical the string.

+

An axis is assessed on a service's own inbound surface, never on its authority over the concept. tenant-engine is the source of existence for tenant records and is nonetheless at I1, because it takes the acting identity from the request body rather than from a verified token. Draft-5 conflated these by naming the authority inside the I2 definition, which made the level describing canonical identity unclaimable by the service that provides it. Corrected on tenant-engine's review — a reader would otherwise assume the authority must be at I2 by definition.

+

business-app-service-contract §2.1 sets app-local accounts as the v1 baseline for business apps — a sanctioned low level with recorded triggers for moving up. That is the pattern this framework generalises.

+

4.2 Authorization (A)

+
+
Authorizationaxis A
A0No authorization, or tenant context not carried.
A1Ad-hoc checks scattered through handlers.
A2A single local authorization boundary; tenant context bound once, centrally.
A3Decisions delegated to flex-auth as PDP, with live re-query where the IAM Profile requires it.
A4A3 over a standard PDP interface (OpenID AuthZEN Authorization API 1.0), so the decision point is swappable and the enforcement point is not coupled to one engine's request shape.
+
+

This ladder describes enforcement points. A decision point cannot occupy A3 — "delegated to flex-auth" is not something flex-auth can do. A service that is a PDP declares two numbers: its own inbound level, and the maximum it enables for consumers. flex-auth reads A0, enables A3 — accurate, and considerably more alarming than A3, which is the point. Raised by flex-auth, whose absence from the §5 worked examples was this surfacing implicitly.

+

A4 is new. The specification reached Final in January 2026 and Keycloak shipped experimental support in May; the argument for it is interoperability — a swappable decision point and an enforcement point not coupled to one engine's request shape.

+

Correction from flex-auth's review: earlier drafts also justified A4 as ending the copying of action strings between repos. It does not. AuthZEN standardises the envelope — subject, action, resource, context, endpoint — and deliberately does not standardise the action vocabulary or the policy language. At A4, tenant.guardrail.set still has to be agreed and still gets copied. Those are two problems with different fixes, and the cheaper one is not A4: flex-auth's registry already carries action definitions per system and could serve them read-only. The vocabulary argument is withdrawn.

+

Internal service-to-service calls are in scope for this axis. "Skipping tenant validation for internal services" is a named anti-pattern and our estate is mostly internal calls.

+

Correction from flex-auth's review: earlier drafts asserted that flex-auth calls tenant-engine synchronously on the authorization path. That is not true. The adapter is built and complete and has no non-test caller, so the IAM Profile's live re-query exists and is unwired — which is also why flex-auth cannot reach I3. Built-and-unwired is the worst of the three states because it reads as capability.

+

The requirement, narrowed on their proposal because the original was too strong to be met and would have made tenant-engine a hard availability dependency of every decision in the estate:

+

Tenant context MUST be carried on every internal hop and MUST NOT be re-derived from a service identity. It MUST be revalidated against tenant-engine at least once per request chain — at the service that holds or mutates the tenant's data, or before a privileged, destructive, credential-vending or aal2-class decision, whichever comes first. A hop that neither holds tenant data nor makes such a decision may carry the context without revalidating it.

+

And carrying tenant context is worthless without an authenticated hop to carry it over. flex-auth found this in itself: it carries tenant context faithfully and cannot distinguish "user-engine asking on behalf of tenant X" from "any pod asking on behalf of tenant X".

+

4.3 Enforcement (E)

+
+
Enforcementaxis E
E0None. Data not tenant-keyed; separation incidental or absent.
E1Data tenant-keyed, filtering applied per query at call sites.
E2Filtering centralised at a single service-side choke point binding authenticated identity to permitted tenants.
E3E2 plus platform-assisted filtering: row-level security keyed on a tenant GUC set transaction-locally, or an equivalent enforced data-access layer.
E4Structural: the credential a workload holds cannot address another tenant's data at all. Requires per-tenant credentials and per-tenant substrate.
+
+

Correction from draft-2. Draft-2 described E3 as something "the application cannot trivially route around". That is false and it was this document overclaiming in exactly the way §6 prohibits. Any session can re-issue SET on a custom GUC, so an attacker with SQL execution can reset the tenant and read across the boundary. What E3 buys is precise, and the ladder must say so:

+
ThreatE1E2E3E4
A developer forgets a tenant predicate
A new code path bypasses the choke point
SQL injection reaching the connection
The application process is compromised
+

E3 is a strong control against accident — the common case, and the one that causes real breaches — and no control at all against compromise. Only E4 holds against both, because the credential itself cannot address another tenant's data.

+

Correction: E3 layers on E2, it does not replace it. External practice treats application-layer and database-layer filtering as complementary. A service that dropped its choke point on reaching E3 would be worse off, since E3 fails open under injection. Claiming E3 therefore requires the E2 evidence artifact as well.

+

Correction: the GUC is set transaction-locally. Draft-2 said "at pool checkout", which is session scope and the wrong instrument. Under a pooler in statement mode, SET leaks between clients and returns other tenants' rows — a failure that appears only under production concurrency and produces no error. Use SET LOCAL inside an explicit transaction.

+

Platform enforcement is a platform obligation. Reaching E3 requires the storage platform to offer the mechanism: provisioned policies, a documented GUC contract, and a probe. Where a consumer wants E3 and the platform has not supplied it, the gap is the platform's. §19.6 asks rapp-postgres to define that contract, which must carry FORCE ROW LEVEL SECURITY on every tenant table (without it the table owner bypasses policies silently, and ADR-0001 already established that our migration role owns the tables it creates), no BYPASSRLS on leased roles, SECURITY INVOKER for ordinary logic, and an EXPLAIN comparison because RLS disables functional indexes built on non-leakproof functions.

+

Not all data is tenant-keyed, and the ladder must not pretend otherwise. A registry whose rows are the tenants has no per-tenant predicate to scope a policy by; enforcing one would break the service's function rather than secure it. tenant-engine's tenants table is the worked example — key-cape enumerates it at token issuance and flex-auth queries it live, both of which are cross-tenant reads by design.

+

A service with mixed data shapes declares E-level plus a registry exception: the level its tenant-keyed tables hold, and a named list of tables excluded because they are registries rather than tenant data. The exception is part of the claim and is reviewable; an unnamed exception is an overclaim. Without this, mixed-shape services either overclaim or stay at E2 permanently, and tenant-engine declined to claim E3 on precisely that reasoning.

+

Default expectation for a new platform service: E2 at first serve, E3 recorded as target. Services whose cross-tenant exposure would be a reportable breach SHOULD target E3 or above.

+

4.4 Placement (P)

+
+
Placementaxis P
P0Shares a database with another consumer.
P1Database per consumer, shared cluster.
P2Dedicated cluster per consumer.
P3Dedicated cluster per tenant.
P4P3 plus separate region or jurisdiction.
+
+

Enforcement and placement are independent axes. Plotted together, with where each service actually sits — parenthesised entries are targets or defaults rather than current positions, and marks a cell the coupling in §3.2 makes unreachable:

+
Enforcement →
E4
business app
E3
target
E2
tenant-engine
audit-core
E1
absorbed repo
E0
P0
P1
P2
P3
P4
Where a service sits todayTarget or defaultUnreachable at this placement
+

P0 → P1 → P2 is movement along the horizontal axis only. Those steps buy consumer isolation, capacity predictability, independent retention and a smaller operational blast radius. They do not raise the tenant boundary by one step. Only P3 makes E4 reachable. This is the most misusable fact in the framework and §11 governs how it may be described.

+

Decision 4.4.1: P1 is the default for platform services; P3 for client-facing business apps, as already ratified. A service unsure which it is must resolve that first (§19.4).

+

Decision 4.4.2 — placement scopes to data substrate. Identity-provider placement (realm-per-tenant versus Organizations) is the same silo/pool decision on a different substrate, is live in our estate, and is undecided. Realm-per-tenant carries a stated ceiling around 5–20 tenants, far below our target. Recorded here as a parallel question (§19.7), not folded into P.

+

4.5 Retention and erasure (R)

+

New in draft-3. Implemented abstractly by the storage platform for any dataset; policy is built on top of that interface by the consumer or its governance layer. Reference implementation: rapp-postgres ADR-0002.

+
+
Retentionaxis R
R0No retention or deletion position. Data kept indefinitely by default; no deletion path exists.
R1Platform default retention applies (N=30 days). The consumer has declared no requirement.
R2Retention declared as N days per dataset; the erasure horizon is published, and the consumer makes no promise shorter than it.
R3Policy-driven deletion: the consumer or its governance layer declares what is due, the platform sweeps whole datasets on that instruction and evidences each run.
R4Verified erasure: data proven unrecoverable across live storage, backups and derived copies, by one of the two routes below.
+
+

R4 has two routes and a service MUST name which one it uses.

+
RouteMechanismCost
Horizon-elapsedWait out the published erasure horizon; the data ages out of every retained copy.Available to everyone, proves little, and the wait is set by a co-resident's retention requirement rather than your own.
Key-destroyedEncrypt per entity, then destroy the key. Retained copies survive but are unreadable.Requires per-entity keys, strong encryption, and an auditable destruction record. Immediate.
+

Decision 4.5.3 — key destruction is not sufficient on its own. The key-destroyed route requires that no retained commitment reveals the erased content. Found by audit-core, and it is a general defect rather than a fact about them:

+
  • A SHA-256 over a canonical record whose fields are low-entropy — event type, actor, tenant, subject, timestamp — is a confirmation oracle. Anyone holding the hash can guess the payload, hash the guess, and confirm a match. Destroying the key does not make the content unrecoverable while that hash survives.
  • Shreddability is not retrofittable onto an integrity chain that commits to cleartext. It has to be built as encrypt-then-hash at accept time, with the chain committing to ciphertext. Retrofitting means rewriting the chain — the exact thing a tamper-evident log exists to make detectable.
+

So a service claiming R4 by key destruction must show that its retained commitments — hashes, chains, indexes, search keys — do not reveal what was erased. The remedies are an HMAC under a per-subject key that dies with the key, or a per-record salt destroyed alongside it. audit-core cannot reach R4 under its current design and targets R2; a fleet R4 target must exempt it explicitly.

+

Regulatory standing of the key-destroyed route, stated carefully because overclaiming here is worse than anywhere else in this framework. Data protection authorities have accepted key destruction as erasure where physical deletion would be manifestly disproportionate, and the practice is recognised under conditions — strong encryption, irreversible destruction, and an auditable record of it. The EDPB has not formally endorsed it as Article 17 erasure. A service reaching R4 by key destruction is making a defensible claim, not a settled one, and must say so rather than reporting a clean "deleted".

+

Three further properties.

+

The erasure horizon is the interval between deleting data and it ceasing to be recoverable from anything the platform holds. Deleting a row does not remove it from yesterday's backup. With an N-day window, deleted data remains recoverable for N days. That is the difference between "deleted" and "erased" and the estate had never written it down.

+

On shared substrate, retention is not per-consumer. Physical backup is instance-wide — one WAL stream, one window — so the instance retention is derived as the maximum across co-resident consumers, and every consumer's horizon is that maximum. A consumer declaring 7 days beside one declaring 90 gets 90. This is the retention analogue of ADR-0001's blast-radius disclosure: state the coupling rather than imply an isolation that is not there.

+

Retention is therefore a placement trigger. A consumer needing a horizon shorter than the instance floor cannot have one at P1. It moves to P2 for a reason with nothing to do with performance — which is exactly why it needs recording, since nobody looks for a retention argument when reviewing placement.

+

Decision 4.5.4 — a retention promise binds both R and P. A tier making a retention claim records an R minimum and a maximum erasure horizon in days. It also requires P2 or above unless its provider contract guarantees that the shared-substrate horizon stays within that maximum and rejects or notifies before a co-resident change would extend it. A bare R2 minimum is insufficient: at P1 another consumer can change the promise without changing the tier or its holder.

+

Deletion splits mechanism from policy. The platform deletes whole datasets on instruction and records an opaque policy reference it never interprets, so every deletion traces to what authorised it. Rows are not a dataset: row expiry is the consumer's own DML under its migration lease. Dropping a consumer's whole database is an operator-gated offboarding step, never a scheduled one.

+

4.6 Availability (V)

+

New in draft-8. adaptive-pricing found that §11 required availability claims to map to a minimum level while the framework supplied no availability vocabulary. Placement is not a substitute: a dedicated cluster can still be a single instance on a single node.

+
+
Availabilityaxis V
V0No availability or recovery position. Recovery is untested or depends on improvisation.
V1Restart or recreate recovery in one failure domain is documented and exercised. Interruption is expected; this is recovery, not failover.
V2Redundant instances provide automated service failover, with measured RTO/RPO; a shared failure domain or critical dependency may remain.
V3The complete critical path survives loss of one declared failure domain, with measured RTO/RPO from an exercise.
V4The complete critical path survives regional loss through tested multi-region failover, with measured RTO/RPO.
+
+

Decision 4.6.1 — V is end-to-end. A service declares the minimum across the components and synchronous providers required to serve the operation. An application with three replicas over a V1 database is V1. A status page or replica count is not evidence of a higher level.

+

Decision 4.6.2 — availability claims name the operation. A read-only degraded mode and a mutation path may have different V levels. Decision 5.2 applies: declare the paths and quote the minimum unless the customer-facing claim explicitly and unambiguously names the narrower operation.

+
+

05The posture vector

+

A service states one level per axis, plus a target, review dates, evidence and any exceptions. current is the highest evidenced level; a control present in code but still awaiting the evidence required by §13 goes in implemented, not in current:

+
schema_version: "0.1"
+framework: netkingdom-tenancy-posture
+service: example-service
+role: tenant-data-service
+tenancy:
+  current:     { I: 2, A: 3, E: 2, P: 1, R: 1, V: 1 }
+  implemented: { E: 3 }
+  target:      { I: 2, A: 3, E: 3, P: 1, R: 2, V: 2 }
+  reviewed: "2026-08-17"
+  review_due: "2027-02-17"
+  service_class: interactive
+  gap:
+    E: "RLS is implemented; the §13 E3 probe is still absent."
+    R: "Retention declared; erasure horizon not yet published to consumers."
+    V: "Automated failover is not implemented or exercised."
+evidence:
+  A3: "docs/evidence/authorization-denial.md"
+  E2: "docs/evidence/cross-tenant-review.md"
+  P1: "rapp-postgres/docs/evidence/isolation-2026-08-10.md"
+

Placement exceptions. Draft-2 assigned one P level per service, which cannot express the vertically partitioned model — most tenants pooled, some dedicated — that §11's isolation tiers require. A tier requiring P2 bought by three tenants would put the service at two levels at once, forcing an over- or under-claim. Placement is therefore declared as a default plus exceptions:

+
  placement_exceptions:
+    - tenants: ["tenant:enterprise:*"]
+      P: 3
+      reason: "isolation tier; see adaptive-pricing tier definition"
+

A service with exceptions must be able to say which tenants are on which substrate. That mapping is a first-class artifact, not archaeology.

+

Decision 5.5 — a provider declares what it makes reachable, not where it sits. The six ladders describe a consumer of infrastructure. They describe a provider of it badly, and railiance-platform's review demonstrated how badly: apps-pg is I0 A0 E0 because a database has no tenant concept, carries no tenant claim and applies no tenant predicate. Those zeros are structural, not weak — the cluster is exactly as strong as its consumers make it.

+

The sharp case is OpenBao at E0. Literally correct, and actively misleading: the mechanism in place is credential-scoped structural separation — E4 machinery — pointed at a consumer boundary rather than a tenant one. A reader scanning a column of E values would rank it below a service doing per-query filtering in application code, inverting the real security position.

+

So a platform service additionally declares, per axis, the level available now, the maximum it can make reachable, and what a consumer must do to reach it. For apps-pg: E4 unreachable (shared credential per consumer, no per-tenant credential), E3 conditional on the GUC contract, R2 blocked on a backup target, V1 at most on the single-node rail. That is the sentence a consumer actually needs, and no arrangement of the consumer ladders produces it.

+

A provider's own P is n/a, not a number. apps-pg provides P1; it is not at P1, and writing P: 1 there would later read as an isolation claim.

+

Worked examples after applying the evidence rule and minimum-across-paths rule consistently:

+
ServiceCurrentNotes
tenant-engineI1 A0 E1 P n/a R0 V0Acting identity is caller-supplied; unauthorised read paths set the A minimum; E2-shaped child-table controls are not evidenced; SQLite is outside P; no erasure or availability evidence. This corrects draft-7, which quoted A2/E2 despite its own minimum/evidence rules.
audit-coreI1 A2 E1 P1 R2 V0E2 is implemented on both paths but awaits the adversarial artifact, so current remains E1. Its 30-day retention and erasure horizon are now declared and published.
flex-authI1 A0 E1 P n/a R n/a V0Enables A3 for consumers. /v1/check authenticates no caller; E2 is implemented but not evidenced.
platform-pg (provider)I0 A0 E0 P n/a R2 V1Provides P1; backup/restore and single-node recovery are evidenced. Provides no tenant boundary by itself.
apps-pg (provider)I0 A0 E0 P n/a R0 V0Zeros are structural, except R0/V0 are live gaps: no backup and no recovery evidence.
adaptive-pricing observatoryI0 A0 E0 P n/a R n/a V0Local, unauthenticated, single-user analysis surface; not a production service.
A newly absorbed repoI1 A1 E1 P0 R0 V0Conformant if declared, with a recorded path.
+

Decision 5.1: the posture vector is declared in the repo, not in the hub, consistent with local-files-are-source-of-truth.

+

Decision 5.2 — declare per path, quote the minimum. A service whose mutations are authorized and whose reads are not is at the reads' level. The quoted number is the minimum across paths; the per-path detail is declared beside it.

+

Draft-6 required only the minimum, on tenant-engine's review. audit-core then showed why that is insufficient on its own: a bare minimum destroys signal, because E3-write/E1-read declares identically to E1/E1. Bare per-path invites "our write path is E3", which is the sentence §6 exists to stop. Both, related explicitly, is the rule.

+

Two services found this shape in themselves within a day of each other — tenant-engine (writes authorized, three read routes not) and audit-core (write path tenant-filtered, read path not filtered at all). Most services enforce harder on write than read, so this is the common case, not the corner.

+

Decision 5.3 — n/a is a level, and it is conformant. P0 presupposes a shared database and R0 presupposes retained data. A service holding nothing at rest — flex-auth runs with its registry and policy baked read-only into the image and no decision log persisted — is neither. A datastore outside a ladder's substrate vocabulary, such as tenant-engine's current SQLite PVC, also uses n/a rather than inventing a level. Without an admissible n/a, a missing rung forces the fabrication §6 prohibits, which is precisely what draft-1 was rejected for. n/a is declared with a stated reason.

+

Decision 5.4 — the vector lives at tenancy.yaml in the repo root. Draft-6 said "in the repo" and not where or in what shape, which left §12's guard needing per-repo archaeology. flex-auth adopted tenancy.yaml speculatively; adopted here as the convention. A repo representing one service uses the single-service form above. A layer repo uses the schema's services list in the same root file — one vector per service, never an average. The normative schema is canon/schemas/tenancy-posture_v0.1.schema.json; prose documents may explain a declaration but do not replace it. The schema carries current, implemented, target, reviewed, review_due, gap, placement_exceptions, service_class (§8.3), per-path detail (§5.2), and provider reachability (§5.5). From the net-kingdom repo, owners validate one or more declarations with uv run tools/tenancy-posture/validate.py <path>...; the validator applies the JSON Schema and the evidence, date, implemented/current and provider-range rules that JSON Schema alone cannot express.

+
+

06Conformance is accuracy, not altitude

+

A service is conformant when its declared posture is accurate, its target is recorded, and it does not claim a level it cannot evidence. It is non-conformant when it overclaims — at any altitude.

+
  • Declaring E0 is conformant. Concealing E0 is not.
  • A repo may be absorbed at any posture. It may not be absorbed silently.
  • No service is blocked from the estate for being low on a ladder. Services MAY be blocked from specific work — serving a tenant grouping, holding a data class, carrying a plan tier — by requirements expressed as minimum levels.
  • Downgrading is permitted and must be declared. A regression found by guarding is a defect; a regression declared in advance is a decision.
  • A low level may be permanent by design, and the declaration must be able to say so. flex-auth is I1 and always will be: a decision point evaluates the claims it is handed, and verifying its own inputs would make it the identity provider its scope refuses to be. A target equal to current with a reason is a settled position, not a stalled trajectory, and §12's guard must not nag it as though it were one.
+

Decision 6.1 — downgrades propagate. Before a planned downgrade of a current level or a provider's available level, the declaring repo MUST resolve the tier definitions and consumers that reference it. A downgrade below a recorded minimum blocks the change until the claim is changed, the workload is moved, or the affected owner explicitly accepts the gap. An unplanned regression is an incident and triggers the same notifications. Updating tenancy.yaml without notifying dependants is declaration drift, not a completed downgrade.

+

Without the axis separation, "not rigorous about tenant separation" is one verdict a repo passes or fails. With it, the same repo is I1 A1 E1 P0 R0 V0 with a path — a plan, not an indictment.

+
+

07Portability across placement levels

+

Movement between P levels must be operational, not a rebuild:

+
  • Connect by injected credential only — no cluster, host, namespace or database name in source.
  • Own a whole database, never tables inside someone else's.
  • Idempotent schema creation.
  • No cross-database joins or co-location assumptions.
+

Decision 7.1: mandatory at P1 and above. At P3, SHOULD rather than MUST — a per-client instance that never moves is not misconformant for naming its own database.

+
+

08Placement triggers

+

Recorded at provisioning time: noisy neighbour on a latency-critical path; a compliance or residency requirement; a plan tier requiring a higher minimum; an erasure horizon that no longer fits (§4.5); connection or memory ceiling reached.

+

Decision 8.1: triggers MUST be monitored, not merely recorded. A trigger in a YAML comment nobody re-reads is documentation, not control.

+

Decision 8.2 — split authority, machine-reconciled. railiance-platform owns the placement rule; the package repo owns the substrate numbers and enforcement; the consuming repo owns its workload requirements; adaptive-pricing owns any tier minimum. adaptive-pricing declined a standing co-signature and the framework accepts the replacement: typed tier minima are joined to consumer and provider declarations at tier definition and whenever one changes. A machine-checkable constraint must not depend on somebody remembering to collect a signature.

+

Decision 8.2.1 — trigger monitoring has an owner. The provider monitors capacity ceilings and co-residency; the consumer monitors latency, compliance and erasure requirements; adaptive-pricing monitors tier-definition changes. The placement owner reconciles those signals. A trigger marked unmonitored is an explicit gap and cannot support a customer assurance claim.

+

8.3 Service class — a placement input, never a priority

+

A latency-critical consumer and a batch consumer can share an instance today with nothing distinguishing them. tenant-engine sits on flex-auth's synchronous authorization path and chose a 5s statement timeout for that reason; audit-core, co-resident, is not latency-critical. Nothing prioritises between them.

+

The framework does not add a QoS axis, because the platform cannot enforce one. Community PostgreSQL has no resource governor: no per-role CPU or I/O priority, no resource queues, no workload classes. Those exist in EDB's enterprise variant, in Greenplum, and in SQL Server — not in what we run. A declared priority level would therefore be an unenforced claim sitting in a declaration, which is precisely what retiring tenantIsolation was about. An axis implies graduation and enforcement; this has neither.

+

Decision 8.3.1 — co-residents are equal. On shared substrate no consumer's query yields to another's. A consumer whose latency requirement cannot survive an unprioritised neighbour must escalate to P2. That is the honest mechanism and it is the only one we have.

+

Decision 8.3.2 — service class is declared anyway, as a category rather than a level: latency-critical, interactive, or batch. It buys three things, none of which is priority:

+
  • A placement input. Mixing latency-critical with batch on one instance is a recognised mismatch. It may still be the right call — it is right today — but it should be a decision, not an accident of who was provisioned when.
  • A trigger. A latency-critical consumer acquiring a batch co-resident is a recorded placement trigger under §8, on the same footing as noisy neighbour.
  • An acceptance criterion for evidence. The noisy-neighbour artifact in §13 asks whether measured degradation is acceptable; without a declared class that word has no referent. Degradation tolerable for batch may be an outage for latency-critical.
+

Decision 8.3.3 — class mixture must be visible. The platform reports which classes are co-resident. An unenforceable risk that nobody can see is strictly worse than one that is stated.

+

The known escalation short of P2 is gateway-level prioritisation — ordering submissions in a connection proxy by the requesting tenant's current consumption. It is real, it is where the industry puts this when it must, and it is new infrastructure we do not run. Recorded as the option, not adopted.

+
+

09Credentials as a tenancy control

+

Short-lived leased credentials re-read at connection checkout, with overlap-first rotation, bound the residual risk at every E level below E4: a leaked credential expires rather than persisting. Stronger than the industry norm of a long-lived per-service secret.

+

Decision 9.1: static long-lived database credentials are not a sanctioned path for any service above E0.

+

Decision 9.2 — the rule extends to consumer-facing credentials. Draft-6 named database access only. audit-core pointed out that its ingest credentials are static long-lived bearer tokens, rotated by publishing a second alongside the first — and that the argument applies with more force to the credential that actually carries the tenant claim than to the one that reaches the database behind it. Read as an accidental omission; it was. Consumer-facing credentials are named in. Where a service cannot yet meet this, it is a stated gap rather than a silent exclusion.

+
+

10Blast radius must be published

+

Decision 10.1: every platform holding consumer data MUST publish, in concrete terms, what a leaked runtime credential can and cannot reach at the levels it operates. rapp-postgres ADR-0001 §5 is the reference. Where the model cannot provide a guarantee, the platform says so and names the escalation.

+

Decision 10.2 — quotas are disclosed, not discovered. The same obligation extends from what a leaked credential can reach to what the platform will refuse to do for you. Every consumer MUST be told, at provisioning, the throttles and quotas enforced against it — connection limits, statement timeouts, idle-transaction timeouts — and told again when they change. A consumer learning its statement timeout by hitting it in production is a disclosure failure, not a consumer bug. This is how tenant-engine was provisioned, by good practice rather than by rule; the rule now exists.

+
+

11Commercial expression

+
  • 11.1 Plan tiers are expressed internally as typed assurance requirements. A tier may require E3 P2 R2 V2 and a maximum erasure horizon; it need not print those labels anywhere customer-facing.
  • 11.2 Marketing and product language is free. No requirement to expose level labels or this document. "Dedicated infrastructure", "isolated tenancy", "private instance" all remain available.
  • 11.3 The constraint is on evidence, not vocabulary. A customer-facing isolation, availability or retention claim must map to a minimum level the delivering service actually holds, recorded once when the tier is defined. The review is internal and happens at tier definition — not per campaign.
  • 11.4 Two hard lines, because these reach contracts and compliance questionnaires:
  • A claim that another tenant cannot reach the customer's data requires E4.
  • A claim that deleted data is gone requires R4, or an erasure horizon disclosed alongside it. Where R4 is reached by key destruction, the claim is defensible but not settled law (§4.5) — it may be made, and it may not be made in language that implies a regulator has blessed it.
  • A claim that service survives loss of a zone requires V3; regional-loss language requires V4. "High availability" without a named failure and measured recovery objective is not an assurance claim this framework can evidence.
  • 11.5 — sanctioned honest language. The strong prohibitions above must not leave a commercial writer with only silence:
  • E3 may be described as database-backed defence against an omitted tenant filter; it must not be paraphrased as "another tenant cannot reach".
  • P2 may be described as a dedicated service database cluster with an independent capacity and restore boundary; it is not tenant-dedicated.
  • R2 may state the declared retention and published erasure horizon.
  • V1 may state exercised restart recovery in one failure domain and must say that interruption and single-domain loss remain.
  • 11.6 — authority and reconciliation. The tier definition is authoritative for the minimum and customer wording. tenancy.yaml is authoritative for the delivering service's current level; provider declarations are authoritative for what infrastructure makes available. None is derived by copying another. Approval joins them and fails closed on a missing, stale or insufficient declaration. A performance-differentiated tier requires P2 or an enforceable resource governor; service class alone grants no priority.
+
+

12Methodology — analyze, establish, improve, guard

+

Analyze. Assess a repo against the ladders; produce tenancy.current with reasoning recorded. Applies to new and absorbed services alike.

+

Establish. Declare the target and gap. The target is set by data class, tenant groupings served and plan tiers carried — not by ambition.

+

Improve. Move one axis at a time. Raising P while leaving E untouched is the characteristic misstep.

+

Guard. Verify continuously that the declared posture holds — against the service's own declaration, not a universal maximum. Nobody must prove every service is at E4; the check is that none is below what it declared.

+

Regression found by guarding is a defect; regression declared in advance is a decision. The estate has been bitten twice by silent pin rollbacks producing ordinary-looking 403s and 404s rather than errors. Posture regression looks the same — an RLS context leak returns correct-looking rows for the wrong tenant. Guarding must be designed for invisible failure, not for crashes.

+
+

13Evidence per level

+

Decision 13.1: a current level is claimed only with its evidence artifact present. This turns §6's accuracy rule from an honour system into a check. implemented records a control observed in code or configuration whose required artifact is still absent; it never satisfies a tier minimum.

+

Decision 13.1a — the floor needs no artifact, only a reason. Found independently by audit-core and flex-auth: the table below defines artifacts from I2, A2, E1, P1, R2 upward and none below, so a literal 13.1 made the lowest rungs unclaimable — including §5's own worked example of a conformant absorbed repo, I1 A1 E1 P0 R0, which could not satisfy it on any axis. A rule that forbids the declaration §6 exists to permit is a defect in the rule.

+

At I0/I1, A0/A1, E0, P0, R0/R1, V0, or n/a, a declaration requires a stated reason, not an artifact. Evidence is what stops you overclaiming, and there is nothing to overclaim at those floors.

+

Decision 13.4 — an artifact must assert something achievable. Draft-3's noisy-neighbour evidence required proof that a saturating consumer "does not breach" another's allowance. Shared infrastructure cannot provide that; the risk is inherent and cannot be wholly removed. An artifact that can only fail, or that passes by being run gently enough, is an overclaim wearing the costume of evidence. Where a property cannot be guaranteed, the artifact measures and records it instead.

+

Decision 13.2 — evidence is of two kinds, and conflating them is an overclaim. Mechanical evidence is a structural assertion a machine can make and belongs in CI. Adversarial evidence is semantic, requires setting up separate tenant contexts and comparing responses, and carries a review date rather than a green build. Cross-tenant findings are the category external testing practice identifies as needing human review. A passing CI run is not E2 evidence.

+
LevelEvidenceKind
I2Identifiers validated against the vocabulary; rejection test for a malformed id; binding shown to come from a verified tokenMechanical
I3Live re-query demonstrated on an aal2-class path; cached-claim path shown unused thereMechanical
A2Choke point identified; test that an unbound request is refusedMechanical
A3Live decision with a denial observed at the endpoint, not only at the decision surfaceMechanical
A4Decision served over the standard interface; a second PDP substituted without PEP change, with the decision differences between the two recorded — substitution proves interface portability, not decision equivalenceMechanical
E1Every tenant-owned table carries the tenant keyMechanical
E2Choke point identified; identity bound to tenant A demonstrably cannot read tenant BAdversarial, with a review date
E3FORCE ROW LEVEL SECURITY on every tenant table; no BYPASSRLS on leased roles; probe that a session without the GUC reads nothing; probe that a wrong GUC reads nothing; EXPLAIN comparisonMechanical
E4Per-tenant credential demonstrated unable to connect to another tenant's substrateMechanical
P1–P4Provisioning declaration plus the platform's isolation probesMechanical
Shared P1–P2 capacity assuranceA recorded baseline of per-consumer resource usage; a run in which one consumer saturates its declared allowance; evidence that the governance controls bind (the greedy consumer is held at its limits) and that the degradation co-residents experience is measured, recorded and judged acceptable against each one's declared service class (§8.3); the aggregate headroom at time of measurementAdversarial, load-generated, with a review date
R2Declared retention rendered; erasure horizon published and reported in the operator surfaceMechanical
R3Sweep evidence records: timestamp, dataset, identifiers removed, authorising policy referenceMechanical
R4Erasure demonstrated across live data, backups and derived copies within the horizonAdversarial
V1Critical dependencies enumerated; restart/recreate recovery exercised; interruption and measured recovery time recordedMechanical exercise
V2One instance terminated while traffic continues or recovers automatically; measured RTO/RPO and remaining shared failure domains recordedAdversarial, failure-injected
V3Declared failure domain removed in an exercise; complete critical path and degraded modes observed against RTO/RPOAdversarial, failure-injected
V4Region made unavailable in an exercise; traffic and state recover in the alternate region against RTO/RPOAdversarial, failure-injected
+

The P1–P4 artifact proves the declared placement topology. The shared-capacity artifact is additional: it is required before a P1/P2 service can claim that a noisy-neighbour control binds, that the trigger is actively guarded, or that a customer performance assurance survives co-residency. It is not required merely to report the true topology as P1 or P2. No such capacity artifact exists in the estate today, so §11 requires P2 or an enforceable governor for a performance-differentiated tier.

+

Decision 13.3: the tenant-boundary E2/E3 and noisy-neighbour artifacts do not exist anywhere in the estate today. rapp-postgres runs 19 adversarial probes, all against the consumer boundary, none against the tenant boundary inside a consumer. Externally, what this framework calls a tenant boundary failure is Broken Object Level Authorization — OWASP API1, top of the API Security Top 10 since that list launched, and the most commonly exploited API vulnerability in published assessments. We have no coverage for the highest-ranked risk in our class of system. §19.3 records the owner.

+
+

14Adoption stance — structure, not tooling

+

Decision 14.1: external research is design input. This estate adopts published standards and structural patterns; it does not adopt tooling unless that tooling is an established industry standard with broad application. Everything else is built ground-up, so it can be optimised and refactored as the estate sees fit.

+
ClassStance
Security baselines (OWASP Multi-Tenant Security Cheat Sheet, API Security Top 10)Adopt as the external reference our ladders answer to
Standards bodies (OpenID AuthZEN 1.0)Adopt — this is what A4 is
Reference taxonomies (Azure tenancy models, AWS SaaS Lens, cell architecture)Adopt as structure
Engine behaviour (PostgreSQL RLS mechanics)Facts, not tooling
Third-party analyzers and test frameworksDo not adopt. Take their rule taxonomies as checklists for probes we write ourselves
+

The practical effect is small and good: rapp-postgres already owns a ground-up probe harness — bash and psql, no dependency tree — that found four real defects in its own provisioning SQL. The evidence artifacts in §13 become new probes in a tool we control. One idea worth reimplementing from the external survey is policy-diff classification: labelling a change to an enforcement policy as safe or breaking before it lands.

+
+

15Alternatives considered

+

One fixed model with a single set of characteristics (draft-1). Rejected: cannot describe a repo that is not there yet, forcing absorbed repos to misrepresent their posture or stay outside. A framework that can only describe its own end state is not a framework.

+

A maturity model with a single overall level. Rejected: collapses the axis separation. A service strong on identity and weak on enforcement has a specific, actionable gap; one composite score hides it and invites averaging.

+

Prohibiting row-level security (draft-2's inherited position). Rejected in draft-2, refined in draft-3: RLS is a real rung against the common threat. The error was never RLS — it was describing E3 in E4's language.

+

Schema-per-consumer in one database. Rejected: pg_catalog is readable per-database, so every co-resident enumerates every other's table and column names regardless of grants. Retained as a describable state, never a target.

+

Mandating E4 for everyone. Rejected: the tenant taxonomy includes consumer (private individuals) and family. A cluster per private individual is economically impossible; the taxonomy is itself evidence pooling is required.

+

Per-consumer physical backup retention. Rejected: CNPG retention is a property of the instance's WAL archive. There is no mechanism, and claiming it would be a fabricated guarantee. Hence the derived maximum in §4.5.

+

Platform-scheduled row expiry. Rejected: requires the platform to hold DML authority over consumer schemas and interpret consumer data semantics, both forbidden by ADR-0001. The consumer's migration lease is the correct instrument.

+

Leaving each repo to its own model. Rejected: the status quo, which produced two contradictory ratified defaults and an unowned placement question.

+
+

16Held against outside practice

+

The graduated reframe is corroborated, not invented here. Microsoft's tenancy-model guidance states it almost verbatim: "Instead of viewing isolation as a discrete property, consider it a spectrum. You can deploy components of your architecture that are more isolated or less isolated than other components in the same architecture." The same guidance derives our E↔P coupling independently — shared deployment means enforcement lives in application code; dedicated deployment means it is structural.

+

Stronger than typical. Most multi-tenancy literature models one boundary, tenant-to-tenant. This estate has two stacked boundaries: platform-service to platform-service, and tenant to tenant inside a consumer. Naming them separately and refusing to enforce both with one mechanism is uncommon and correct. Graduated per-axis levels also beat the silo/pool/bridge trichotomy, which is approximately our P axis with the other four missing — which is why it cannot express "pooled infrastructure, structurally enforced boundary".

+

Weaker than typical. The pool model's standard mitigation is a verified enforcement layer every service is demonstrably routed through. We have the concept and none of the verification (§13.3).

+

Adopted without naming it. Short-lived leased credentials re-read at checkout beat the long-lived-secret norm. §9 promotes it to a tenancy control.

+

Still unexplored. Neither P nor R describes a cell — a slice of infrastructure with a fixed maximum size, sized so one cell's failure is survivable and cell count scales linearly. platform-pg is, in these terms, an uncapped cell: §17 computes a ceiling and nothing enforces it (§19.8).

+

Sources: the five research digests in the-custodian/research/2026-08-17-adr008-*, which carry full citations for the claims in this section.

+
+

17Scaling demands

+

Derived from the live platform-pg specification. Connection arithmetic is exact; the per-backend memory estimate remains unmeasured and is explicitly a gap in rapp-postgres ADR-0004.

+
instances:        1              (no HA; single-node rail)
+max_connections:  100
+memory limit:     1Gi
+per consumer:     14 connections (12 runtime + 2 migration)
+

The hard connection bound is roughly six declarations; the enforceable operational ceiling is four. Seven declarations request 98 of 100 connections before CNPG's instance manager, metrics exporter and reserved slots. Every one of them is politely inside its declared 14-connection allowance; the instance still fails. ADR-0004 sets four because memory is expected to bind first and fails by OOM-killing every co-resident rather than refusing one connection.

+

That distinction matters because our governance addresses the wrong shape. Per-consumer connection_limit, statement_timeout and idle_in_transaction_session_timeout guard well against one greedy consumer. They do nothing about the aggregate of many modest ones, which is the second and less intuitive noisy-neighbour failure and the one this number describes. Two workload consumers plus the isolation probe occupy three of the four declared slots. The next workload request must trigger measurement and the overflow decision before admission.

+

Memory likely binds first. 100 backends against 1Gi is ~10MB per backend. Connection exhaustion errors clearly; memory pressure OOM-kills and degrades every co-resident at once.

+

E3 and pooling. Corrected from draft-2, which had this backwards. Transaction-scoped context (SET LOCAL inside an explicit transaction) is what makes E3 safe under a pooler. Statement-level pooling is what breaks it, serving other tenants' rows under concurrency with no error. E3 constrains which pooling mode is available, not whether pooling is available.

+

Retention consumes the volume. WAL accumulates with the window, and §4.5 makes the window the maximum across consumers. A consumer declaring a long retention extends everyone's horizon and everyone's storage draw against a 20Gi volume.

+

Restore time couples all consumers. Physical backup is instance-wide, so a consumer's RTO is a function of total instance size, not its own.

+

platform-pg is V1. instances: 1 on a single-node rail provides exercised restart recovery and no failover. P1 describes its consumer placement and says nothing about this availability fact; the new V axis carries it.

+
+

18Consequences

+
  • The estate gains one vocabulary and a way to be honest about partial adoption.
  • Absorbed repos get a described state and a path instead of a failing grade.
  • tenantIsolation in PostgresConsumer is revealed as a mislabelled field.
  • The verification problem becomes tractable: guard against declaration.
  • Draft-2's RLS prohibition is reversed and its E3 description corrected; rapp-postgres acquires an obligation to define and offer the mechanism.
  • Adding a consumer with long retention silently extends everyone's erasure horizon. This must reach the consumer review checklist, not only this document.
  • A service selling an isolation tier must maintain a tenant→substrate mapping it does not have today.
  • Availability becomes an end-to-end, evidenced property rather than an inference from replica count or placement.
  • Nothing here changes a running system.
+
+

19Review resolutions and residual questions

+
  1. tenantIsolation field — resolved. rapp-postgres retired it. A consumer declaration asks for mechanisms; posture lives in the consumer's tenancy.yaml.
  2. Placement ownership — resolved. §8.2 records the split. The policy has one owner; typed tier requirements replace the declined commercial co-signature.
  3. E2, E3 and noisy-neighbour evidenceowned as of 2026-08-17 by whitehat-security (WHITEHAT-WP-0001), an independent adversarial evidence facility seeded for this purpose. audit-core and tenant-engine were right to decline it as fleet-scope work; the answer was a home of its own rather than a volunteer.
+

Owned by NetKingdom — corrected 2026-08-17; an earlier revision of this section proposed otherwise on independence grounds and was overruled. Offensive security is security work and belongs with the repo that owns security. The facility is framed offensively rather than as a conformance checker: it is pointed at infrastructure we choose, our own estate among them, and conformance testing is one use of a general capability.

+

The residual tension is recorded rather than resolved: NetKingdom owns this framework and the facility that tests conformance to it, so those findings are NetKingdom assessing NetKingdom. The mitigation is that findings leave for risk-nexus, under the-custodian, rather than being closed in place. Proportionate, not perfect. Revisit if conformance findings start getting quietly closed.

+

Two consequences land back here. Cadence is now a security parameter, not a schedule — for any control whose guarantee is detection rather than prevention, the interval between probe runs is the exposure window, and rapp-postgres ADR-0003 leaves that number to the facility. And a passing suite is not proof of isolation; it is proof that the attacks attempted did not work. §13's evidence artifacts should be read with that distinction, because a green run recorded as "E2 verified" would be exactly the overclaim §6 prohibits.

+
  1. Business app vs platform service — open. Custodian canon: a classification rule. Candidate: reuse repo-classification-standard_v1.0.
  2. Tier → minimum level mapping — policy resolved, implementation open. adaptive-pricing owns typed minima and wording; tenant-engine owns plan assignment by id. Current tiers make no assurance claims.
  3. The E3 mechanism — resolved. rapp-postgres ADR-0003 publishes the GUC contract with the FORCE/BYPASSRLS/SECURITY INVOKER/EXPLAIN requirements.
  4. Identity-provider placement — open. Owner of key-cape: realm-per-tenant or Organizations? Realm-per-tenant's ~5–20 tenant ceiling is below our target.
  5. Cell sizing — resolved for platform-pg. rapp-postgres ADR-0004 sets four consumers and names absent overflow target platform-pg-2; measurement and provisioning remain live gaps.
  6. Retention floor and ceiling — resolved as policy. Both exist; requests outside them fail validation and the package repo owns the numbers.
  7. Engine neutrality — open. The P ladder rests on a PostgreSQL property. State it engine-specifically and say so, or abstract it and risk a non-Postgres implementation that silently differs?
  8. Erasure versus audit — framework resolved. audit-core: crypto-shredding a tenant's audit records destroys the evidence the service exists to hold, and ADR-0001 §2 deliberately built the role model so history could not be rewritten. The usual resolution separates the fact of an event, retained, from its personal payload, encrypted per subject and shreddable. Raised because a naive "R4 everywhere" target would instruct the audit service to destroy its own evidence. audit-core targets R2 and is explicitly not a fleet R4 target. The legal basis for retaining audit facts remains a risk/legal question outside this framework.
  9. Quality of serviceresolved 2026-08-17. Co-residents are equal; a declared service class informs placement but never grants priority. See §8.3. The question asked whether to add a QoS dimension; the answer is no, and the reason is that we could not enforce one.
+

Routed elsewhere, deliberately. The tenant identifier tenant:<grouping>:<name> embeds headcount bands (small, medium, large) that change as a tenant grows, contradicting the consensus that identifiers should not encode mutable attributes. That is a critique of ADR-0013, not of this framework, and belongs to tenant-engine and NetKingdom canon. Folding it in here would overreach.

+
+

20Ratification path

+
  1. Reviewed by tenant-engine, flex-auth, audit-core, rapp-postgres, railiance-platform and adaptive-pricing against §19. Complete in draft-8.
  2. Each publishes its own posture vector (§5) as part of review. The framework is validated by whether it can describe them accurately — if a repo cannot express itself in these six ladders, the ladders are wrong and this document changes, not the repo. Complete in draft-8; all six root declarations validate against the canonical schema.
  3. On acceptance, supersedes the routing of rapp-postgres/docs/canon-drafts/shared-platform-relational-storage_v0.1-draft.md, whose §§3–8 are absorbed here. That draft is withdrawn rather than left pending.
  4. On acceptance, rapp-postgres ADR-0001 through ADR-0004 move to accepted and are annotated as the PostgreSQL implementation of the E, P, R and shared-capacity rules.
+
netkingdom-tenancy-posture · draft-8 · proposednet-kingdom · canon/standards/tenancy-posture_v0.1.md · cced59d3aa1dc0aa08fc128fc8c76699f59dcd90
diff --git a/build/tenancy-posture.html b/build/tenancy-posture.html index 776bb14..b5906e9 100644 --- a/build/tenancy-posture.html +++ b/build/tenancy-posture.html @@ -1,383 +1 @@ -Tenancy Posture - -
netkingdom-tenancy-posture proposed · draft-5generated from canon — do not edit

Tenancy Posture

A framework for describing, holding and improving multi-tenancy — including where we are not there yet.

Status

-

Proposed, draft-5. Relocated from the-custodian/canon/architecture on 2026-08-17: multi-tenancy is part of the IT-security framework NetKingdom provides, so this framework belongs in NetKingdom canon beside the IAM Profile and the tenant-engine boundary contract, not in the work-factory canon.

-
  • draft-1 proposed a single model with fixed characteristics. Rejected: it could not describe a repo that is not there yet.
  • draft-2 reframed to graduated levels per axis. Externally corroborated (§16), but four of its statements were wrong and one thing it needed was missing.
  • draft-3 applied those corrections, added the retention axis, and recorded an adoption stance.
  • draft-4 closed the two gaps draft-3 left open: R4 had no mechanism beyond waiting, and the noisy-neighbour evidence artifact asserted something shared infrastructure cannot provide.
  • draft-5 relocates to NetKingdom and renames the dimensions from planes to axes, because the word was already taken (§0).
-

Every correction so far was found by research or by relocation, not by review.

-

Informed by five external research digests in research/2026-08-17-adr008-*, which carry full citations for every external claim made here.

-

Reviewed by nobody yet. §19 lists what each owner is being asked to accept.

-
-

00Terminology: axes, not planes

-

docs/platform-identity-security-architecture.md — accepted, 2026-07-23 — already uses plane for a trust and deployment layer: the bootstrap plane, the platform control plane, and tenant planes. That meaning is established, ratified, and owned by this repo.

-

Drafts 1–4 of this document, written elsewhere, used plane for something different: an independent dimension of concern. Two incompatible senses of one word inside one canon is exactly the concept-ownership collision the estate has been careful about elsewhere, and the newcomer yields.

-

This framework therefore describes five axes. They are orthogonal to NetKingdom's planes, not a subdivision of them:

-
  • A plane is where something runs and what trust it carries — bootstrap, platform control, tenant.
  • An axis is which property of tenancy is being described — identity, authorization, enforcement, placement, retention.
-

A workload in the tenant plane has a position on all five axes. A platform control plane service does too. The two vocabularies compose and neither replaces the other.

-

The rename is also an improvement. A posture vector is literally a point in five-dimensional space, and "axis" says that where "plane" did not.

-
-

01Context

-

Drafts 1–4 opened by claiming the estate "has never written down what it is building". Relocation proved that wrong, and the correction is worth keeping visible: docs/platform-identity-security-architecture.md has described the trust model, the tenant model and a capability progression since 2026-07-23. The accurate claim is narrower — what was missing is a way to say how far a given service has got, and to hold several answers at once. Seven documents cover slices of the subject and none of them does that:

-
DocumentCoversStatus
iam-profile_v0.3 (NetKingdom)Tenant identifier shape, tenant_roles claim, staleness rulesRatified
tenant-engine-boundary-contract_v0.1 (NetKingdom)Who owns tenant records, roles, plan assignmentRatified
business-app-service-contract_v0.1 §1 (Custodian)Business apps: instance-per-client, tenant-keyed dataRatified
rapp-postgres ADR-0001Consumer + tenant isolation in PostgreSQLProposed, governs one repo
rapp-postgres ADR-0002Per-consumer retention and the erasure horizonProposed, governs one repo
shared-platform-relational-storage_v0.1The stacked-boundary gapRouted 2026-08-10, still unratified
platform-identity-security-architecture (NetKingdom)Trust model, planes, tenant model, capability progressionAccepted 2026-07-23
-

This document is downstream of that architecture and must not restate it. It answers one question the architecture leaves open: given the model, where is this particular service today, and how would anyone know?

-

Four failures follow.

-

The gap was diagnosed once and the fix stalled. The v0.1 draft was written to fill this hole and has sat unratified in neither canon directory. §20 attaches a ratification path so this one does not join it.

-

Placement is owned by nobody. user-engine-pg and target-revenue-pg are dedicated; apps-pg, net-kingdom-pg, platform-pg, state-hub-db and forgejo-db are shared. Both live, neither written down. tenant-engine raised this with railiance-platform on 2026-08-16; unanswered.

-

Two contradictory defaults are already ratified. Business apps get instance-per-client; platform services pool. Nothing says which shape a new service takes, and no definition separates the categories.

-

There is no honest way to describe a repo that is not there yet. The estate absorbs repos with weak or absent tenant separation. Today such a repo is simply non-conformant, leaving it two bad options: misrepresent its posture, or stay outside the framework.

-
-

02What this document is

-

A framework, not a model. It specifies no single correct implementation. It supplies terminology (§3, §4), a declaration (§5), a conformance rule (§6), methodology (§12), and evidence definitions (§13).

-

A service is conformant when its declared posture is accurate and its trajectory recorded. A service is non-conformant when it claims a level it cannot evidence — regardless of how high or low that level is.

-
-

03Five orthogonal axes

-

"Is this multi-tenant?" is treated as one question. It is five, and they are independent:

-
AxisQuestionVocabulary owner
Identity (I)How is a tenant named and validated?tenant-engine / IAM Profile
Authorization (A)How is a request bound to the tenants it may act for?flex-auth
Enforcement (E)Where, mechanically, is the tenant boundary enforced?This framework
Placement (P)Which substrate holds a tenant's data?railiance-platform
Retention (R)How long does data persist, and how is it erased?The storage platform; policy by the consumer
-

Conflation produces errors today. rapp-postgres's PostgresConsumer carries tenantIsolation: consumer-service-boundary — an E-axis fact in a P-axis artifact, reading as though storage enforces something it does not. The "dedicated versus shared" argument mixes P (capacity, blast radius) with E (correctness).

-

The axes are separated precisely so each may sit at a different level.

-

Decision 3.1: every document, declaration and plan tier that says "isolation" MUST name which axis it means.

-

Decision 3.2: the axes couple at their tops and the couplings MUST be stated where they apply, not used to argue the axes are one:

-
  • E4 is reachable only at P3 or above.
  • R's erasure horizon is bounded below by P — on shared substrate, a consumer's horizon is the instance maximum (§4.5).
  • R4 by key destruction is bounded by the key boundary, which is an E-axis property. Shredding a single tenant's data requires the application to encrypt under a per-tenant key before writing; the storage platform cannot supply it. Reaching the top of the retention ladder is not a retention project.
-

Decision 3.3 — scope. The P and R ladders describe a service's primary datastore. Caches, search indices, message queues and background jobs are named leak surfaces in the external baselines and are assessed separately, not covered by a posture vector. Saying so is honest; implying the vector covers them would not be.

-
-

04Graduated levels

-

Each axis carries an ordered ladder. Higher is stronger, not better: the right level is the one a service can evidence and its risk warrants.

-

4.1 Identity (I)

-
-
Identityplane I
I0No tenant concept. Data not attributable to a tenant.
I1A local tenant notion exists but is not canonical, or the tenant is taken from the request rather than from a verified token.
I2Canonical identifiers, bound at the identity provider and carried as a verified claim; tenant-engine is the source of existence.
I3I2 plus capability roles honoured, with live tenant-engine re-query for privileged, destructive, credential-vending or aal2-class decisions.
Ladder ends at I3.
-
-

I1 now explicitly absorbs request-supplied tenant identifiers. "Never trust client-supplied tenant IDs without validation" is a named anti-pattern; a service reading the tenant from a header is at I1 however canonical the string.

-

business-app-service-contract §2.1 sets app-local accounts as the v1 baseline for business apps — a sanctioned low level with recorded triggers for moving up. That is the pattern this framework generalises.

-

4.2 Authorization (A)

-
-
Authorizationplane A
A0No authorization, or tenant context not carried.
A1Ad-hoc checks scattered through handlers.
A2A single local authorization boundary; tenant context bound once, centrally.
A3Decisions delegated to flex-auth as PDP, with live re-query where the IAM Profile requires it.
A4A3 over a standard PDP interface (OpenID AuthZEN Authorization API 1.0), so the decision point is swappable and the enforcement point is not coupled to one engine's request shape.
-
-

A4 is new. flex-auth uses a bespoke CheckRequest and a bespoke action vocabulary, with action strings copied verbatim between repos to avoid re-derivation — exactly the coupling AuthZEN removes. The specification reached Final in January 2026 and Keycloak shipped experimental support in May. We are not wrong, we are pre-standard, and the ladder should have somewhere to go.

-

Internal service-to-service calls are in scope for this axis. "Skipping tenant validation for internal services" is a named anti-pattern, and our estate is mostly internal calls — flex-auth calls tenant-engine synchronously on the authorization path. A service identity acting on behalf of a tenant must carry and revalidate tenant context to claim A2 or above.

-

4.3 Enforcement (E)

-
-
Enforcementplane E
E0None. Data not tenant-keyed; separation incidental or absent.
E1Data tenant-keyed, filtering applied per query at call sites.
E2Filtering centralised at a single service-side choke point binding authenticated identity to permitted tenants.
E3E2 plus platform-assisted filtering: row-level security keyed on a tenant GUC set transaction-locally, or an equivalent enforced data-access layer.
E4Structural: the credential a workload holds cannot address another tenant's data at all. Requires per-tenant credentials and per-tenant substrate.
-
-

Correction from draft-2. Draft-2 described E3 as something "the application cannot trivially route around". That is false and it was this document overclaiming in exactly the way §6 prohibits. Any session can re-issue SET on a custom GUC, so an attacker with SQL execution can reset the tenant and read across the boundary. What E3 buys is precise, and the ladder must say so:

-
ThreatE1E2E3E4
A developer forgets a tenant predicate
A new code path bypasses the choke point
SQL injection reaching the connection
The application process is compromised
-

E3 is a strong control against accident — the common case, and the one that causes real breaches — and no control at all against compromise. Only E4 holds against both, because the credential itself cannot address another tenant's data.

-

Correction: E3 layers on E2, it does not replace it. External practice treats application-layer and database-layer filtering as complementary. A service that dropped its choke point on reaching E3 would be worse off, since E3 fails open under injection. Claiming E3 therefore requires the E2 evidence artifact as well.

-

Correction: the GUC is set transaction-locally. Draft-2 said "at pool checkout", which is session scope and the wrong instrument. Under a pooler in statement mode, SET leaks between clients and returns other tenants' rows — a failure that appears only under production concurrency and produces no error. Use SET LOCAL inside an explicit transaction.

-

Platform enforcement is a platform obligation. Reaching E3 requires the storage platform to offer the mechanism: provisioned policies, a documented GUC contract, and a probe. Where a consumer wants E3 and the platform has not supplied it, the gap is the platform's. §19.6 asks rapp-postgres to define that contract, which must carry FORCE ROW LEVEL SECURITY on every tenant table (without it the table owner bypasses policies silently, and ADR-0001 already established that our migration role owns the tables it creates), no BYPASSRLS on leased roles, SECURITY INVOKER for ordinary logic, and an EXPLAIN comparison because RLS disables functional indexes built on non-leakproof functions.

-

Default expectation for a new platform service: E2 at first serve, E3 recorded as target. Services whose cross-tenant exposure would be a reportable breach SHOULD target E3 or above.

-

4.4 Placement (P)

-
-
Placementplane P
P0Shares a database with another consumer.
P1Database per consumer, shared cluster.
P2Dedicated cluster per consumer.
P3Dedicated cluster per tenant.
P4P3 plus separate region or jurisdiction.
-
-

Enforcement and placement are independent axes. Plotted together, with where each service actually sits — parenthesised entries are targets or defaults rather than current positions, and marks a cell the coupling in §3.2 makes unreachable:

-
Enforcement →
E4
business app
E3
target
E2
tenant-engineaudit-core
E1
absorbed repo
E0
P0
P1
P2
P3
P4
Where a service sits todayTarget or defaultUnreachable at this placement
-

P0 → P1 → P2 is movement along the horizontal axis only. Those steps buy consumer isolation, capacity predictability, independent retention and a smaller operational blast radius. They do not raise the tenant boundary by one step. Only P3 makes E4 reachable. This is the most misusable fact in the framework and §11 governs how it may be described.

-

Decision 4.4.1: P1 is the default for platform services; P3 for client-facing business apps, as already ratified. A service unsure which it is must resolve that first (§19.4).

-

Decision 4.4.2 — placement scopes to data substrate. Identity-provider placement (realm-per-tenant versus Organizations) is the same silo/pool decision on a different substrate, is live in our estate, and is undecided. Realm-per-tenant carries a stated ceiling around 5–20 tenants, far below our target. Recorded here as a parallel question (§19.7), not folded into P.

-

4.5 Retention and erasure (R)

-

New in draft-3. Implemented abstractly by the storage platform for any dataset; policy is built on top of that interface by the consumer or its governance layer. Reference implementation: rapp-postgres ADR-0002.

-
-
Retentionplane R
R0No retention or deletion position. Data kept indefinitely by default; no deletion path exists.
R1Platform default retention applies (N=30 days). The consumer has declared no requirement.
R2Retention declared as N days per dataset; the erasure horizon is published, and the consumer makes no promise shorter than it.
R3Policy-driven deletion: the consumer or its governance layer declares what is due, the platform sweeps whole datasets on that instruction and evidences each run.
R4Verified erasure: data proven unrecoverable across live storage, backups and derived copies, by one of the two routes below.
-
-

R4 has two routes and a service MUST name which one it uses.

-
RouteMechanismCost
Horizon-elapsedWait out the published erasure horizon; the data ages out of every retained copy.Available to everyone, proves little, and the wait is set by a co-resident's retention requirement rather than your own.
Key-destroyedEncrypt per entity, then destroy the key. Retained copies survive but are unreadable.Requires per-entity keys, strong encryption, and an auditable destruction record. Immediate.
-

Regulatory standing of the key-destroyed route, stated carefully because overclaiming here is worse than anywhere else in this framework. Data protection authorities have accepted key destruction as erasure where physical deletion would be manifestly disproportionate, and the practice is recognised under conditions — strong encryption, irreversible destruction, and an auditable record of it. The EDPB has not formally endorsed it as Article 17 erasure. A service reaching R4 by key destruction is making a defensible claim, not a settled one, and must say so rather than reporting a clean "deleted".

-

Three further properties.

-

The erasure horizon is the interval between deleting data and it ceasing to be recoverable from anything the platform holds. Deleting a row does not remove it from yesterday's backup. With an N-day window, deleted data remains recoverable for N days. That is the difference between "deleted" and "erased" and the estate had never written it down.

-

On shared substrate, retention is not per-consumer. Physical backup is instance-wide — one WAL stream, one window — so the instance retention is derived as the maximum across co-resident consumers, and every consumer's horizon is that maximum. A consumer declaring 7 days beside one declaring 90 gets 90. This is the retention analogue of ADR-0001's blast-radius disclosure: state the coupling rather than imply an isolation that is not there.

-

Retention is therefore a placement trigger. A consumer needing a horizon shorter than the instance floor cannot have one at P1. It moves to P2 for a reason with nothing to do with performance — which is exactly why it needs recording, since nobody looks for a retention argument when reviewing placement.

-

Deletion splits mechanism from policy. The platform deletes whole datasets on instruction and records an opaque policy reference it never interprets, so every deletion traces to what authorised it. Rows are not a dataset: row expiry is the consumer's own DML under its migration lease. Dropping a consumer's whole database is an operator-gated offboarding step, never a scheduled one.

-
-

05The posture vector

-

A service states one level per axis, plus a target, a date, and any placement exceptions:

-
tenancy:
-  current:  { I: 2, A: 3, E: 2, P: 1, R: 1 }
-  target:   { I: 2, A: 3, E: 3, P: 1, R: 2 }
-  reviewed: "2026-08-17"
-  gap:
-    E: "Choke point exists and is tested; RLS not provisioned. Blocked on
-        rapp-postgres publishing the GUC contract. Target Q4."
-    R: "Retention declared; erasure horizon not yet published to consumers."
-

Placement exceptions. Draft-2 assigned one P level per service, which cannot express the vertically partitioned model — most tenants pooled, some dedicated — that §11's isolation tiers require. A tier requiring P2 bought by three tenants would put the service at two levels at once, forcing an over- or under-claim. Placement is therefore declared as a default plus exceptions:

-
  placement_exceptions:
-    - tenants: ["tenant:enterprise:*"]
-      P: 3
-      reason: "isolation tier; see adaptive-pricing tier definition"
-

A service with exceptions must be able to say which tenants are on which substrate. That mapping is a first-class artifact, not archaeology.

-

Worked examples, best-effort and subject to owner correction:

-
ServiceCurrentNotes
tenant-engineI2 A3 E2 P1 R1Moving to P1 under TEN-WP-0009; retention declared, horizon not yet published.
audit-coreI2 A3 E2 P1 R1Holds audit evidence, so both E3 and R2 are urgent targets.
A newly absorbed repoI1 A1 E1 P0 R0Conformant if declared, with a recorded path.
-

Decision 5.1: the posture vector is declared in the repo, not in the hub, consistent with local-files-are-source-of-truth.

-
-

06Conformance is accuracy, not altitude

-

A service is conformant when its declared posture is accurate, its target is recorded, and it does not claim a level it cannot evidence. It is non-conformant when it overclaims — at any altitude.

-
  • Declaring E0 is conformant. Concealing E0 is not.
  • A repo may be absorbed at any posture. It may not be absorbed silently.
  • No service is blocked from the estate for being low on a ladder. Services MAY be blocked from specific work — serving a tenant grouping, holding a data class, carrying a plan tier — by requirements expressed as minimum levels.
  • Downgrading is permitted and must be declared. A regression found by guarding is a defect; a regression declared in advance is a decision.
-

Without the axis separation, "not rigorous about tenant separation" is one verdict a repo passes or fails. With it, the same repo is I1 A1 E1 P0 R0 with a path — a plan, not an indictment.

-
-

07Portability across placement levels

-

Movement between P levels must be operational, not a rebuild:

-
  • Connect by injected credential only — no cluster, host, namespace or database name in source.
  • Own a whole database, never tables inside someone else's.
  • Idempotent schema creation.
  • No cross-database joins or co-location assumptions.
-

Decision 7.1: mandatory at P1 and above. At P3, SHOULD rather than MUST — a per-client instance that never moves is not misconformant for naming its own database.

-
-

08Placement triggers

-

Recorded at provisioning time: noisy neighbour on a latency-critical path; a compliance or residency requirement; a plan tier requiring a higher minimum; an erasure horizon that no longer fits (§4.5); connection or memory ceiling reached.

-

Decision 8.1: triggers MUST be monitored, not merely recorded. A trigger in a YAML comment nobody re-reads is documentation, not control.

-

Decision 8.2: placement policy ownership is proposed to railiance-platform, co-signed by adaptive-pricing. Tenancy model selection is a commercial decision as much as a technical one; an operations-shaped repo should not hold it alone.

-
-

09Credentials as a tenancy control

-

Short-lived leased credentials re-read at connection checkout, with overlap-first rotation, bound the residual risk at every E level below E4: a leaked credential expires rather than persisting. Stronger than the industry norm of a long-lived per-service secret.

-

Decision 9.1: static long-lived database credentials are not a sanctioned path for any service above E0.

-
-

10Blast radius must be published

-

Decision 10.1: every platform holding consumer data MUST publish, in concrete terms, what a leaked runtime credential can and cannot reach at the levels it operates. rapp-postgres ADR-0001 §5 is the reference. Where the model cannot provide a guarantee, the platform says so and names the escalation.

-

Decision 10.2 — quotas are disclosed, not discovered. The same obligation extends from what a leaked credential can reach to what the platform will refuse to do for you. Every consumer MUST be told, at provisioning, the throttles and quotas enforced against it — connection limits, statement timeouts, idle-transaction timeouts — and told again when they change. A consumer learning its statement timeout by hitting it in production is a disclosure failure, not a consumer bug. This is how tenant-engine was provisioned, by good practice rather than by rule; the rule now exists.

-
-

11Commercial expression

-
  • 11.1 Plan tiers are expressed internally as minimum levels. A tier may require E3 P2 R2; it need not print that anywhere customer-facing.
  • 11.2 Marketing and product language is free. No requirement to expose level labels or this document. "Dedicated infrastructure", "isolated tenancy", "private instance" all remain available.
  • 11.3 The constraint is on evidence, not vocabulary. A customer-facing isolation, availability or retention claim must map to a minimum level the delivering service actually holds, recorded once when the tier is defined. The review is internal and happens at tier definition — not per campaign.
  • 11.4 Two hard lines, because these reach contracts and compliance questionnaires:
  • A claim that another tenant cannot reach the customer's data requires E4.
  • A claim that deleted data is gone requires R4, or an erasure horizon disclosed alongside it. Where R4 is reached by key destruction, the claim is defensible but not settled law (§4.5) — it may be made, and it may not be made in language that implies a regulator has blessed it.
-
-

12Methodology — analyze, establish, improve, guard

-

Analyze. Assess a repo against the ladders; produce tenancy.current with reasoning recorded. Applies to new and absorbed services alike.

-

Establish. Declare the target and gap. The target is set by data class, tenant groupings served and plan tiers carried — not by ambition.

-

Improve. Move one axis at a time. Raising P while leaving E untouched is the characteristic misstep.

-

Guard. Verify continuously that the declared posture holds — against the service's own declaration, not a universal maximum. Nobody must prove every service is at E4; the check is that none is below what it declared.

-

Regression found by guarding is a defect; regression declared in advance is a decision. The estate has been bitten twice by silent pin rollbacks producing ordinary-looking 403s and 404s rather than errors. Posture regression looks the same — an RLS context leak returns correct-looking rows for the wrong tenant. Guarding must be designed for invisible failure, not for crashes.

-
-

13Evidence per level

-

Decision 13.1: a level is claimed only with its evidence artifact present. This turns §6's accuracy rule from an honour system into a check.

-

Decision 13.4 — an artifact must assert something achievable. Draft-3's noisy-neighbour evidence required proof that a saturating consumer "does not breach" another's allowance. Shared infrastructure cannot provide that; the risk is inherent and cannot be wholly removed. An artifact that can only fail, or that passes by being run gently enough, is an overclaim wearing the costume of evidence. Where a property cannot be guaranteed, the artifact measures and records it instead.

-

Decision 13.2 — evidence is of two kinds, and conflating them is an overclaim. Mechanical evidence is a structural assertion a machine can make and belongs in CI. Adversarial evidence is semantic, requires setting up separate tenant contexts and comparing responses, and carries a review date rather than a green build. Cross-tenant findings are the category external testing practice identifies as needing human review. A passing CI run is not E2 evidence.

-
LevelEvidenceKind
I2Identifiers validated against the vocabulary; rejection test for a malformed id; binding shown to come from a verified tokenMechanical
I3Live re-query demonstrated on an aal2-class path; cached-claim path shown unused thereMechanical
A2Choke point identified; test that an unbound request is refusedMechanical
A3Live decision with a denial observed at the endpoint, not only at the decision surfaceMechanical
A4Decision served over the standard interface; a second PDP substituted without PEP changeMechanical
E1Every tenant-owned table carries the tenant keyMechanical
E2Choke point identified; identity bound to tenant A demonstrably cannot read tenant BAdversarial, with a review date
E3FORCE ROW LEVEL SECURITY on every tenant table; no BYPASSRLS on leased roles; probe that a session without the GUC reads nothing; probe that a wrong GUC reads nothing; EXPLAIN comparisonMechanical
E4Per-tenant credential demonstrated unable to connect to another tenant's substrateMechanical
P1–P4Provisioning declaration plus the platform's isolation probesMechanical
P1–P2 (noisy neighbour)A recorded baseline of per-consumer resource usage; a run in which one consumer saturates its declared allowance; evidence that the governance controls bind (the greedy consumer is held at its limits) and that the degradation co-residents experience is measured, recorded and judged acceptable; the aggregate headroom at time of measurementAdversarial, load-generated, with a review date
R2Declared retention rendered; erasure horizon published and reported in the operator surfaceMechanical
R3Sweep evidence records: timestamp, dataset, identifiers removed, authorising policy referenceMechanical
R4Erasure demonstrated across live data, backups and derived copies within the horizonAdversarial
-

Decision 13.3: the E2, E3 and noisy-neighbour artifacts do not exist anywhere in the estate today. rapp-postgres runs 15 adversarial probes, all against the consumer boundary, none against the tenant boundary inside a consumer. Externally, what this framework calls a tenant boundary failure is Broken Object Level Authorization — OWASP API1, top of the API Security Top 10 since that list launched, and the most commonly exploited API vulnerability in published assessments. We have no coverage for the highest-ranked risk in our class of system. §19.3 seeks an owner.

-
-

14Adoption stance — structure, not tooling

-

Decision 14.1: external research is design input. This estate adopts published standards and structural patterns; it does not adopt tooling unless that tooling is an established industry standard with broad application. Everything else is built ground-up, so it can be optimised and refactored as the estate sees fit.

-
ClassStance
Security baselines (OWASP Multi-Tenant Security Cheat Sheet, API Security Top 10)Adopt as the external reference our ladders answer to
Standards bodies (OpenID AuthZEN 1.0)Adopt — this is what A4 is
Reference taxonomies (Azure tenancy models, AWS SaaS Lens, cell architecture)Adopt as structure
Engine behaviour (PostgreSQL RLS mechanics)Facts, not tooling
Third-party analyzers and test frameworksDo not adopt. Take their rule taxonomies as checklists for probes we write ourselves
-

The practical effect is small and good: rapp-postgres already owns a ground-up probe harness — bash and psql, no dependency tree — that found four real defects in its own provisioning SQL. The evidence artifacts in §13 become new probes in a tool we control. One idea worth reimplementing from the external survey is policy-diff classification: labelling a change to an enforcement policy as safe or breaking before it lands.

-
-

15Alternatives considered

-

One fixed model with a single set of characteristics (draft-1). Rejected: cannot describe a repo that is not there yet, forcing absorbed repos to misrepresent their posture or stay outside. A framework that can only describe its own end state is not a framework.

-

A maturity model with a single overall level. Rejected: collapses the axis separation. A service strong on identity and weak on enforcement has a specific, actionable gap; one composite score hides it and invites averaging.

-

Prohibiting row-level security (draft-2's inherited position). Rejected in draft-2, refined in draft-3: RLS is a real rung against the common threat. The error was never RLS — it was describing E3 in E4's language.

-

Schema-per-consumer in one database. Rejected: pg_catalog is readable per-database, so every co-resident enumerates every other's table and column names regardless of grants. Retained as a describable state, never a target.

-

Mandating E4 for everyone. Rejected: the tenant taxonomy includes consumer (private individuals) and family. A cluster per private individual is economically impossible; the taxonomy is itself evidence pooling is required.

-

Per-consumer physical backup retention. Rejected: CNPG retention is a property of the instance's WAL archive. There is no mechanism, and claiming it would be a fabricated guarantee. Hence the derived maximum in §4.5.

-

Platform-scheduled row expiry. Rejected: requires the platform to hold DML authority over consumer schemas and interpret consumer data semantics, both forbidden by ADR-0001. The consumer's migration lease is the correct instrument.

-

Leaving each repo to its own model. Rejected: the status quo, which produced two contradictory ratified defaults and an unowned placement question.

-
-

16Held against outside practice

-

The graduated reframe is corroborated, not invented here. Microsoft's tenancy-model guidance states it almost verbatim: "Instead of viewing isolation as a discrete property, consider it a spectrum. You can deploy components of your architecture that are more isolated or less isolated than other components in the same architecture." The same guidance derives our E↔P coupling independently — shared deployment means enforcement lives in application code; dedicated deployment means it is structural.

-

Stronger than typical. Most multi-tenancy literature models one boundary, tenant-to-tenant. This estate has two stacked boundaries: platform-service to platform-service, and tenant to tenant inside a consumer. Naming them separately and refusing to enforce both with one mechanism is uncommon and correct. Graduated per-axis levels also beat the silo/pool/bridge trichotomy, which is approximately our P axis with the other four missing — which is why it cannot express "pooled infrastructure, structurally enforced boundary".

-

Weaker than typical. The pool model's standard mitigation is a verified enforcement layer every service is demonstrably routed through. We have the concept and none of the verification (§13.3).

-

Adopted without naming it. Short-lived leased credentials re-read at checkout beat the long-lived-secret norm. §9 promotes it to a tenancy control.

-

Still unexplored. Neither P nor R describes a cell — a slice of infrastructure with a fixed maximum size, sized so one cell's failure is survivable and cell count scales linearly. platform-pg is, in these terms, an uncapped cell: §17 computes a ceiling and nothing enforces it (§19.8).

-

Sources: the four research digests in research/2026-08-17-adr008-*, which carry full citations for every claim in this section.

-
-

17Scaling demands

-

Measured against the live platform-pg specification, not estimated.

-
instances:        1              (no HA; single-node rail)
-max_connections:  100
-memory limit:     1Gi
-per consumer:     14 connections (12 runtime + 2 migration)
-

Connection ceiling: roughly six consumers — and this is the aggregate noisy-neighbour bound, not a capacity statistic. Seven consumers request 98 of 100 before CNPG's instance manager, metrics exporter and reserved slots. Every one of them is politely inside its declared 14-connection allowance; the instance still fails.

-

That distinction matters because our governance addresses the wrong shape. Per-consumer connection_limit, statement_timeout and idle_in_transaction_session_timeout guard well against one greedy consumer. They do nothing about the aggregate of many modest ones, which is the second and less intuitive noisy-neighbour failure and the one this number describes. Two consumers are provisioned. We are at roughly a third of the bound, and the third request will not feel like a scaling event.

-

Memory likely binds first. 100 backends against 1Gi is ~10MB per backend. Connection exhaustion errors clearly; memory pressure OOM-kills and degrades every co-resident at once.

-

E3 and pooling. Corrected from draft-2, which had this backwards. Transaction-scoped context (SET LOCAL inside an explicit transaction) is what makes E3 safe under a pooler. Statement-level pooling is what breaks it, serving other tenants' rows under concurrency with no error. E3 constrains which pooling mode is available, not whether pooling is available.

-

Retention consumes the volume. WAL accumulates with the window, and §4.5 makes the window the maximum across consumers. A consumer declaring a long retention extends everyone's horizon and everyone's storage draw against a 20Gi volume.

-

Restore time couples all consumers. Physical backup is instance-wide, so a consumer's RTO is a function of total instance size, not its own.

-

No P1 tenant has HA. instances: 1 means a tier promising uptime cannot be satisfied at P1 as built — an availability floor belongs in §11's minimum-level vocabulary alongside isolation.

-
-

18Consequences

-
  • The estate gains one vocabulary and a way to be honest about partial adoption.
  • Absorbed repos get a described state and a path instead of a failing grade.
  • tenantIsolation in PostgresConsumer is revealed as a mislabelled field.
  • The verification problem becomes tractable: guard against declaration.
  • Draft-2's RLS prohibition is reversed and its E3 description corrected; rapp-postgres acquires an obligation to define and offer the mechanism.
  • Adding a consumer with long retention silently extends everyone's erasure horizon. This must reach the consumer review checklist, not only this document.
  • A service selling an isolation tier must maintain a tenant→substrate mapping it does not have today.
  • Nothing here changes a running system.
-
-

19Open questions

-
  1. tenantIsolation fieldrapp-postgres: rename to name its axis and carry a level (tenancy.E: 2), or move it out of the storage declaration.
  2. Placement ownershiprailiance-platform with adaptive-pricing: accept the ladder, triggers and the §8.1 monitoring obligation; appoint a recorded placement owner per workload.
  3. E2, E3 and noisy-neighbour evidenceowner needed. Now three artifacts of two kinds: E3 and noisy-neighbour are buildable as probes in the existing harness; E2 is adversarial and needs a review cadence. Both audit-core and tenant-engine have declined fleet-scope work on correct boundary reasoning, so this needs appointing. Highest-severity gap.
  4. Business app vs platform service — Custodian canon: a classification rule. Candidate: reuse repo-classification-standard_v1.0.
  5. Tier → minimum level mappingadaptive-pricing and tenant-engine: required only for tiers making isolation, availability or retention claims.
  6. The E3 mechanismrapp-postgres: publish the GUC contract with the FORCE/BYPASSRLS/SECURITY INVOKER/EXPLAIN requirements in §4.3.
  7. Identity-provider placement — owner of key-cape: realm-per-tenant or Organizations? Realm-per-tenant's ~5–20 tenant ceiling is below our target.
  8. Cell sizing — reframed from "should we adopt cells" to "what is platform-pg's declared maximum size, and what is the overflow target?" The connection ceiling forces this whether or not we adopt the vocabulary.
  9. Retention floor and ceiling — should backupRetentionDays have a platform minimum (so a consumer asking for 1 day gets a validation error rather than a quiet disappointment) and a maximum (so nobody exhausts the volume)?
  10. Engine neutrality — the P ladder rests on a PostgreSQL property. State it engine-specifically and say so, or abstract it and risk a non-Postgres implementation that silently differs?
  11. Erasure versus auditaudit-core: crypto-shredding a tenant's audit records destroys the evidence the service exists to hold, and ADR-0001 §2 deliberately built the role model so history could not be rewritten. The usual resolution separates the fact of an event, retained, from its personal payload, encrypted per subject and shreddable. Raised because a naive "R4 everywhere" target would instruct the audit service to destroy its own evidence. The answer is audit-core's, not this framework's.
  12. Quality of serviceowner needed. The framework has no vocabulary for saying one consumer's latency matters more than another's. tenant-engine sits on flex-auth's synchronous authorization path and chose a 5s statement timeout for that reason; it shares an instance with audit-core, which is not latency-critical. Nothing prioritises between them. Either add a QoS dimension or state that all co-residents are equal and latency-critical consumers must escalate to P2.
-

Routed elsewhere, deliberately. The tenant identifier tenant:<grouping>:<name> embeds headcount bands (small, medium, large) that change as a tenant grows, contradicting the consensus that identifiers should not encode mutable attributes. That is a critique of ADR-0013, not of this framework, and belongs to tenant-engine and NetKingdom canon. Folding it in here would overreach.

-
-

20Ratification path

-
  1. Reviewed by tenant-engine, flex-auth, rapp-postgres, railiance-platform and adaptive-pricing against §19.
  2. Each publishes its own posture vector (§5) as part of review. The framework is validated by whether it can describe them accurately — if a repo cannot express itself in these five ladders, the ladders are wrong and this document changes, not the repo.
  3. On acceptance, supersedes the routing of rapp-postgres/docs/canon-drafts/shared-platform-relational-storage_v0.1-draft.md, whose §§3–8 are absorbed here. That draft is withdrawn rather than left pending.
  4. On acceptance, rapp-postgres ADR-0001 and ADR-0002 move to accepted and are annotated as the PostgreSQL implementation of the E, P and R ladders.
-
netkingdom-tenancy-posture · draft-5 · proposedgenerated from the-custodian/canon
+NetKingdom Tenancy Posture v0.1

Moved permanently to /standards/tenancy-posture/v0.1/index.html.

diff --git a/deploy/nginx.conf b/deploy/nginx.conf new file mode 100644 index 0000000..1b7839d --- /dev/null +++ b/deploy/nginx.conf @@ -0,0 +1,38 @@ +map $uri $policy_cache_control { + default "no-cache"; + ~^/standards/.+/revisions/ "public, max-age=31536000, immutable"; +} + +server { + listen 8080 default_server; + listen [::]:8080 default_server; + server_name _; + server_tokens off; + absolute_redirect off; + + root /usr/share/nginx/html; + index index.html; + charset utf-8; + etag on; + + add_header Cache-Control $policy_cache_control always; + add_header Content-Security-Policy "default-src 'none'; style-src 'unsafe-inline'; img-src 'self' data:; font-src 'self'; base-uri 'none'; form-action 'none'; frame-ancestors 'none'" always; + add_header Permissions-Policy "camera=(), geolocation=(), microphone=()" always; + add_header Referrer-Policy "no-referrer" always; + add_header X-Content-Type-Options "nosniff" always; + add_header X-Frame-Options "DENY" always; + + location = /healthz { + access_log off; + default_type text/plain; + return 200 "ok\n"; + } + + location = /tenancy-posture.html { + return 308 /standards/tenancy-posture/v0.1/index.html; + } + + location / { + try_files $uri $uri/ $uri/index.html =404; + } +} diff --git a/docs/adr/ADR-0001-addressing-and-permanence.md b/docs/adr/ADR-0001-addressing-and-permanence.md new file mode 100644 index 0000000..bae45d2 --- /dev/null +++ b/docs/adr/ADR-0001-addressing-and-permanence.md @@ -0,0 +1,60 @@ +# ADR-0001 — policy addressing and permanence + +- Status: accepted +- Date: 2026-08-18 +- Owner: the-custodian + +## Decision + +A document has one stable current address and immutable revision addresses: + +```text +//// +////revisions// +``` + +The publication manifest records both. Existing public paths become permanent +redirect aliases; the first is `/tenancy-posture.html`. No URL is derived from +a checkout path, branch name, build number or hosting implementation. + +The current address advances only when the owning source publishes a new +revision. A revision address is write-once: the builder records the exact git +commit (or commit plus working-tree marker) and the full source-content digest. +It refuses to replace that revision with different content. Committing the +same bytes may update the current page's provenance but does not rewrite the +already-published revision page. The generated publication manifest makes the +source repository, path, revision and digest machine-readable. + +Superseded and withdrawn documents are never deleted. The current page gains a +status banner and link to its successor; every historical revision continues +to resolve. “Withdrawn” means retained and visibly non-current. Deletion is not +a lifecycle state. + +Source Git history remains the authority for versions that predate this site. +Once a revision is published, the built revision page is also retained by the +hosting artifact/release. Rollback republishes a previous complete build; it +does not rebuild old pages from a changed working tree. + +## Permanence promise + +A published URL is expected to resolve indefinitely. Breaking that promise +requires an explicit the-custodian decision plus a redirect/export plan. A DNS, +hosting or repository migration does not qualify: those must preserve paths. + +This repo owns addressing, rendering and currency. It does not decide whether +a draft is accepted; ratification belongs to the canon owner/the-custodian. + +## Publication scope + +Canon `standards`, `architecture`, and `constitution` are in scope. `values`, +`tpsc`, and `projects` are out until their owner marks individual documents as +governing. Per-repo ADRs are in scope through explicit manifest entries. +Workplans, evidence, runbooks and general docs are not. + +## Consequences + +- Builds fail if a source disappears, an id differs, a path collides, or an + immutable revision would change; stale output is not silently called fresh. +- Pages show status, revision, owner, last review and exact source revision. +- Availability remains restart recovery on the single-node rail. This contract + promises stable addressing, not a high-availability SLA. diff --git a/publication.json b/publication.json new file mode 100644 index 0000000..0bb177e --- /dev/null +++ b/publication.json @@ -0,0 +1,30 @@ +{ + "schema_version": 1, + "site": { + "title": "Coulomb Policy Nexus", + "base_url": "https://policy.coulomb.social" + }, + "canon_scope": { + "standards": true, + "architecture": true, + "constitution": true, + "values": false, + "tpsc": false, + "projects": false + }, + "repositories": { + "net-kingdom": {"path": "../net-kingdom"} + }, + "documents": [ + { + "id": "netkingdom-tenancy-posture", + "source_repo": "net-kingdom", + "source_path": "canon/standards/tenancy-posture_v0.1.md", + "canonical_path": "standards/tenancy-posture/v0.1/index.html", + "revision_path": "standards/tenancy-posture/v0.1/revisions/{revision}/index.html", + "legacy_paths": ["tenancy-posture.html"], + "subtitle": "A framework for describing, holding and improving multi-tenancy — including where we are not there yet.", + "review_interval": "6m" + } + ] +} diff --git a/tests/test_publication.py b/tests/test_publication.py new file mode 100644 index 0000000..7a33ca8 --- /dev/null +++ b/tests/test_publication.py @@ -0,0 +1,197 @@ +from __future__ import annotations + +import importlib.util +import json +import datetime as dt +import os +from pathlib import Path +import sys +import tempfile +import unittest +from unittest import mock + + +ROOT = Path(__file__).parents[1] +sys.path.insert(0, str(ROOT / "tools")) +SPEC = importlib.util.spec_from_file_location("build_site", ROOT / "tools/build_site.py") +build_site = importlib.util.module_from_spec(SPEC) +assert SPEC.loader is not None +SPEC.loader.exec_module(build_site) + + +def _fixture(tmp_path: Path) -> Path: + repo = tmp_path / "canon-repo" + source = repo / "canon/standards/example.md" + source.parent.mkdir(parents=True) + source.write_text( + """--- +id: example +title: "Example Standard" +status: proposed +revision: "draft-1" +owner: example-owner +last_reviewed: "2026-08-18" +review_interval: 6m +--- + +# Example + +## 1. Rule + +| Level | Meaning | +| --- | --- | +| **V0** | No position. | +| **V1** | Restart recovery. | +""", + encoding="utf-8", + ) + manifest = tmp_path / "publication.json" + manifest.write_text( + json.dumps( + { + "schema_version": 1, + "site": {"title": "Test", "base_url": "https://example.invalid"}, + "repositories": {"canon": {"path": "canon-repo"}}, + "documents": [ + { + "id": "example", + "source_repo": "canon", + "source_path": "canon/standards/example.md", + "canonical_path": "standards/example/v1/index.html", + "revision_path": "standards/example/v1/revisions/{revision}/index.html", + "legacy_paths": ["example.html"], + "review_interval": "6m", + } + ], + } + ), + encoding="utf-8", + ) + return manifest + + +class PublicationTest(unittest.TestCase): + def test_archive_build_accepts_only_exact_source_revision_override(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + repo = root / "net-kingdom" + source = repo / "canon/example.md" + source.parent.mkdir(parents=True) + source.write_text("example", encoding="utf-8") + key = "POLICY_NEXUS_SOURCE_REVISION_NET_KINGDOM" + with mock.patch.dict(os.environ, {key: "a" * 40}): + self.assertEqual("a" * 40, build_site._source_revision(repo, source)) + with mock.patch.dict(os.environ, {key: "main"}): + with self.assertRaisesRegex(ValueError, "clean 40-hex Git commit"): + build_site._source_revision(repo, source) + + def test_manifest_builds_index_current_revision_and_legacy_alias(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + manifest = _fixture(root) + output = root / "site" + records = build_site.build(manifest, output, as_of=dt.date(2026, 8, 18)) + + self.assertEqual("2027-02-18", records[0]["review_due"]) + self.assertTrue((output / "index.html").is_file()) + current = (output / "standards/example/v1/index.html").read_text( + encoding="utf-8" + ) + revision = output / "standards/example/v1/revisions/draft-1/index.html" + self.assertIn("Availability", current) + self.assertIn("policy-source-revision", current) + self.assertIn("policy-source-digest", current) + self.assertIn("Review due: 2027-02-18", current) + self.assertEqual(revision.read_text(encoding="utf-8"), current) + self.assertIn( + "/standards/example/v1/index.html", + (output / "example.html").read_text(encoding="utf-8"), + ) + + def test_revision_address_refuses_changed_source_under_same_revision(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + manifest = _fixture(root) + output = root / "site" + build_site.build(manifest, output, as_of=dt.date(2026, 8, 18)) + source = root / "canon-repo/canon/standards/example.md" + source.write_text( + source.read_text(encoding="utf-8") + "\nChanged.\n", encoding="utf-8" + ) + + with self.assertRaisesRegex(RuntimeError, "immutable revision"): + build_site.build(manifest, output, as_of=dt.date(2026, 8, 18)) + + def test_same_content_can_move_from_worktree_to_commit_provenance(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + manifest = _fixture(root) + output = root / "site" + build_site.build(manifest, output, as_of=dt.date(2026, 8, 18)) + revision = output / "standards/example/v1/revisions/draft-1/index.html" + original = revision.read_text(encoding="utf-8") + + original_source_revision = build_site._source_revision + try: + build_site._source_revision = lambda _repo, _source: "new-commit" + build_site.build(manifest, output, as_of=dt.date(2026, 8, 18)) + finally: + build_site._source_revision = original_source_revision + + self.assertEqual(original, revision.read_text(encoding="utf-8")) + current = (output / "standards/example/v1/index.html").read_text( + encoding="utf-8" + ) + self.assertIn('policy-source-revision" content="new-commit', current) + + def test_stale_and_superseded_notices_do_not_mutate_revision_page(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + manifest = _fixture(root) + document = json.loads(manifest.read_text(encoding="utf-8")) + document["documents"][0]["lifecycle"] = "superseded" + document["documents"][0]["successor"] = "/standards/example/v2/" + manifest.write_text(json.dumps(document), encoding="utf-8") + output = root / "site" + + build_site.build(manifest, output, as_of=dt.date(2027, 2, 19)) + + current = (output / "standards/example/v1/index.html").read_text( + encoding="utf-8" + ) + revision = ( + output / "standards/example/v1/revisions/draft-1/index.html" + ).read_text(encoding="utf-8") + self.assertIn("Superseded.", current) + self.assertIn("Review overdue.", current) + self.assertNotIn("Superseded.", revision) + self.assertNotIn("Review overdue.", revision) + + def test_manifest_rejects_unsafe_publication_path(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + manifest = _fixture(root) + document = json.loads(manifest.read_text(encoding="utf-8")) + document["documents"][0]["canonical_path"] = "../escape.html" + manifest.write_text(json.dumps(document), encoding="utf-8") + + with self.assertRaisesRegex(ValueError, "unsafe publication path"): + build_site.load_manifest(manifest) + + def test_build_rejects_missing_publication_owner(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + manifest = _fixture(root) + source = root / "canon-repo/canon/standards/example.md" + source.write_text( + source.read_text(encoding="utf-8").replace( + "owner: example-owner\n", "" + ), + encoding="utf-8", + ) + with self.assertRaisesRegex(ValueError, "owner is required"): + build_site.build(manifest, root / "site") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_release.py b/tests/test_release.py new file mode 100644 index 0000000..9ba7b7f --- /dev/null +++ b/tests/test_release.py @@ -0,0 +1,106 @@ +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +import sys +import tempfile +import unittest + +ROOT = Path(__file__).parents[1] +sys.path.insert(0, str(ROOT / "tools")) +from verify_release import verify + + +COMMIT = "1" * 40 +SOURCE_DIGEST = "2" * 64 + + +def _page(revision: str = COMMIT, digest: str = SOURCE_DIGEST) -> str: + return ( + '' + ) + + +def _release(root: Path) -> Path: + build = root / "build" + canonical = build / "standards/example/v1/index.html" + revision = build / "standards/example/v1/revisions/draft-1/index.html" + canonical.parent.mkdir(parents=True) + revision.parent.mkdir(parents=True) + canonical.write_text(_page(), encoding="utf-8") + revision.write_text(_page(), encoding="utf-8") + (build / "index.html").write_text("", encoding="utf-8") + manifest = { + "schema_version": 1, + "generated_as_of": "2026-08-18", + "documents": [ + { + "id": "example", + "title": "Example Standard", + "status": "proposed", + "revision": "draft-1", + "owner": "example-owner", + "last_reviewed": "2026-08-18", + "review_due": "2027-02-18", + "canonical_path": "standards/example/v1/index.html", + "revision_path": "standards/example/v1/revisions/draft-1/index.html", + "source_revision": COMMIT, + "source_digest": SOURCE_DIGEST, + } + ], + } + (build / "publication-manifest.json").write_text( + json.dumps(manifest, sort_keys=True) + "\n", encoding="utf-8" + ) + return build + + +class ReleaseVerificationTest(unittest.TestCase): + def test_accepts_clean_provenance_and_returns_manifest_digest(self) -> None: + with tempfile.TemporaryDirectory() as directory: + build = _release(Path(directory)) + evidence = verify(build) + expected = hashlib.sha256( + (build / "publication-manifest.json").read_bytes() + ).hexdigest() + self.assertEqual(expected, evidence["publication_manifest_digest"]) + self.assertEqual(["example"], evidence["documents"]) + + def test_rejects_working_tree_source_revision(self) -> None: + with tempfile.TemporaryDirectory() as directory: + build = _release(Path(directory)) + manifest_path = build / "publication-manifest.json" + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + dirty = COMMIT + "+working-tree." + SOURCE_DIGEST[:12] + manifest["documents"][0]["source_revision"] = dirty + manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + canonical = build / manifest["documents"][0]["canonical_path"] + canonical.write_text(_page(revision=dirty), encoding="utf-8") + + with self.assertRaisesRegex(ValueError, "clean 40-hex Git commit"): + verify(build) + + def test_rejects_immutable_revision_built_from_dirty_source(self) -> None: + with tempfile.TemporaryDirectory() as directory: + build = _release(Path(directory)) + revision = build / "standards/example/v1/revisions/draft-1/index.html" + revision.write_text( + _page(revision=COMMIT + "+working-tree." + SOURCE_DIGEST[:12]), + encoding="utf-8", + ) + with self.assertRaisesRegex(ValueError, "immutable revision page"): + verify(build) + + def test_rejects_page_digest_mismatch(self) -> None: + with tempfile.TemporaryDirectory() as directory: + build = _release(Path(directory)) + canonical = build / "standards/example/v1/index.html" + canonical.write_text(_page(digest="3" * 64), encoding="utf-8") + with self.assertRaisesRegex(ValueError, "provenance differs"): + verify(build) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/build_site.py b/tools/build_site.py new file mode 100644 index 0000000..2e3089b --- /dev/null +++ b/tools/build_site.py @@ -0,0 +1,300 @@ +#!/usr/bin/env python3 +"""Build the policy site atomically from an explicit source manifest.""" + +from __future__ import annotations + +import argparse +import datetime as dt +import hashlib +import html +import json +import os +from pathlib import Path, PurePosixPath +import re +import shutil +import subprocess +import tempfile +from typing import Any + +from render import STYLE, render_page, split_frontmatter + + +SOURCE_REVISION = re.compile( + r'' +) +SOURCE_DIGEST = re.compile(r'') +CLEAN_GIT_REVISION = re.compile(r"^[a-f0-9]{40}$") + + +def _safe_path(value: str) -> PurePosixPath: + path = PurePosixPath(value) + if path.is_absolute() or ".." in path.parts or not path.parts: + raise ValueError(f"unsafe publication path: {value!r}") + return path + + +def _source_revision(repo: Path, source: Path) -> str: + digest = hashlib.sha256(source.read_bytes()).hexdigest() + revision_env = "POLICY_NEXUS_SOURCE_REVISION_" + re.sub( + r"[^A-Z0-9]+", "_", repo.name.upper() + ) + supplied_revision = os.environ.get(revision_env, "") + if supplied_revision: + if not CLEAN_GIT_REVISION.fullmatch(supplied_revision): + raise ValueError( + f"{revision_env} must be a clean 40-hex Git commit, got {supplied_revision!r}" + ) + return supplied_revision + try: + head = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + relative = source.relative_to(repo) + dirty = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain", "--", str(relative)], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + return f"{head}+working-tree.{digest[:12]}" if dirty else head + except (OSError, subprocess.CalledProcessError, ValueError): + return f"sha256:{digest}" + + +def _add_interval(reviewed: str, interval: str) -> dt.date: + date = dt.date.fromisoformat(reviewed) + match = re.fullmatch(r"([1-9][0-9]*)([dmy])", interval) + if not match: + raise ValueError(f"invalid review interval {interval!r}; expected Nd, Nm or Ny") + amount, unit = int(match.group(1)), match.group(2) + if unit == "d": + return date + dt.timedelta(days=amount) + months = amount * (12 if unit == "y" else 1) + month_index = date.month - 1 + months + year, month = date.year + month_index // 12, month_index % 12 + 1 + month_lengths = (31, 29 if year % 4 == 0 and (year % 100 != 0 or year % 400 == 0) else 28, + 31, 30, 31, 30, 31, 31, 30, 31, 30, 31) + return date.replace(year=year, month=month, day=min(date.day, month_lengths[month - 1])) + + +def load_manifest(path: Path) -> dict[str, Any]: + manifest = json.loads(path.read_text(encoding="utf-8")) + if manifest.get("schema_version") != 1: + raise ValueError("publication manifest schema_version must be 1") + if not manifest.get("documents"): + raise ValueError("publication manifest has no documents") + seen: set[PurePosixPath] = set() + for document in manifest["documents"]: + for raw in ( + document["canonical_path"], + document["revision_path"].replace("{revision}", "revision"), + *document.get("legacy_paths", []), + ): + path_value = _safe_path(raw) + if path_value in seen: + raise ValueError(f"duplicate publication path: {path_value}") + seen.add(path_value) + return manifest + + +def _redirect(target: str, title: str) -> str: + escaped = html.escape(target, quote=True) + return ( + "" + f'' + f"{html.escape(title)}" + f'

Moved permanently to {escaped}.

\n' + ) + + +def _index_page(site: dict[str, Any], records: list[dict[str, str]]) -> str: + rows = [] + for record in records: + rows.append( + "" + f'' + f'{html.escape(record["title"])}' + f'{html.escape(record["status"])}' + f'{html.escape(record["lifecycle"])}' + f'{html.escape(record["revision"])}' + f'{html.escape(record["owner"])}' + f'{html.escape(record["last_reviewed"])}' + f'{html.escape(record["review_due"])}' + f'{html.escape(record["currency"])}' + "" + ) + return ( + "" + f"{html.escape(site['title'])}" + '
policy surface' + "generated from canonical sources — do not edit
" + f"

{html.escape(site['title'])}

" + '

Canon and architecture decisions at stable addresses, with visible currency.

' + "
" + "" + "" + f"{''.join(rows)}
DocumentStatusLifecycleRevisionOwnerReviewedReview dueCurrency
\n" + ) + + +def build( + manifest_path: Path, + output: Path, + *, + as_of: dt.date | None = None, +) -> list[dict[str, str]]: + manifest_path = manifest_path.resolve() + manifest = load_manifest(manifest_path) + as_of = as_of or dt.date.today() + repository_paths = { + name: (manifest_path.parent / config["path"]).resolve() + for name, config in manifest["repositories"].items() + } + output_parent = output.resolve().parent + output_parent.mkdir(parents=True, exist_ok=True) + temporary = Path(tempfile.mkdtemp(prefix=f".{output.name}-", dir=output_parent)) + if output.exists(): + shutil.copytree(output, temporary, dirs_exist_ok=True) + + records: list[dict[str, str]] = [] + try: + for document in manifest["documents"]: + repo = repository_paths[document["source_repo"]] + source = (repo / document["source_path"]).resolve() + if not source.is_file() or repo not in source.parents: + raise FileNotFoundError(f"canonical source unavailable: {source}") + meta, _markdown = split_frontmatter(source.read_text(encoding="utf-8")) + if meta.get("id") != document["id"]: + raise ValueError( + f"{source}: manifest id {document['id']!r} does not match {meta.get('id')!r}" + ) + for required_field in ("title", "status", "owner"): + if not meta.get(required_field): + raise ValueError(f"{source}: {required_field} is required for publication") + revision = meta.get("revision") or meta.get("version") + if not revision: + raise ValueError(f"{source}: revision or version is required") + source_revision = _source_revision(repo, source) + source_digest = hashlib.sha256(source.read_bytes()).hexdigest() + reviewed = meta.get("last_reviewed") or meta.get("updated") + interval = document.get("review_interval") or meta.get("review_interval") + if not reviewed or not interval: + raise ValueError(f"{source}: review date and interval are required") + review_due = _add_interval(reviewed, interval) + lifecycle = document.get("lifecycle", "active") + if lifecycle not in {"active", "superseded", "withdrawn"}: + raise ValueError( + f"{document['id']}: lifecycle must be active, superseded or withdrawn" + ) + successor = document.get("successor", "") + if lifecycle == "superseded" and not successor: + raise ValueError(f"{document['id']}: superseded documents require successor") + publication = { + "source_repo": document["source_repo"], + "source_path": document["source_path"], + "source_revision": source_revision, + "source_digest": source_digest, + "review_due": review_due.isoformat() if review_due else "", + } + revision_page, meta, _sections = render_page( + source, + subtitle=document.get("subtitle", ""), + publication=publication, + ) + current_publication = publication | { + "lifecycle": lifecycle, + "successor": successor, + "stale": "true" if review_due and review_due < as_of else "false", + } + current_page, _current_meta, _current_sections = render_page( + source, + subtitle=document.get("subtitle", ""), + publication=current_publication, + ) + canonical = _safe_path(document["canonical_path"]) + revision_path = _safe_path(document["revision_path"].format(revision=revision)) + revision_target = temporary / revision_path + if revision_target.exists(): + existing_revision = revision_target.read_text(encoding="utf-8") + old_revision = SOURCE_REVISION.search(existing_revision) + old_digest = SOURCE_DIGEST.search(existing_revision) + if not old_revision or not old_digest: + raise RuntimeError( + f"immutable revision {revision_path} has incomplete source metadata" + ) + if html.unescape(old_digest.group(1)) != source_digest: + raise RuntimeError( + f"immutable revision {revision_path} already records content digest " + f"{old_digest.group(1)}; source is now {source_digest}. " + "Publish a new revision id." + ) + canonical_target = temporary / canonical + canonical_target.parent.mkdir(parents=True, exist_ok=True) + canonical_target.write_text(current_page, encoding="utf-8") + if not revision_target.exists(): + revision_target.parent.mkdir(parents=True, exist_ok=True) + revision_target.write_text(revision_page, encoding="utf-8") + for legacy in document.get("legacy_paths", []): + target = temporary / _safe_path(legacy) + target.parent.mkdir(parents=True, exist_ok=True) + canonical_url = "/" + canonical.as_posix() + target.write_text(_redirect(canonical_url, meta["title"]), encoding="utf-8") + records.append( + { + "id": document["id"], + "title": meta["title"], + "status": meta["status"], + "revision": revision, + "owner": meta["owner"], + "last_reviewed": reviewed, + "review_due": review_due.isoformat(), + "currency": "stale" if review_due < as_of else "current", + "lifecycle": lifecycle, + "canonical_path": canonical.as_posix(), + "revision_path": revision_path.as_posix(), + "source_revision": source_revision, + "source_digest": source_digest, + } + ) + + (temporary / "index.html").write_text( + _index_page(manifest["site"], records), encoding="utf-8" + ) + (temporary / "publication-manifest.json").write_text( + json.dumps( + { + "schema_version": 1, + "generated_as_of": as_of.isoformat(), + "documents": records, + }, + indent=2, + sort_keys=True, + ) + + "\n", + encoding="utf-8", + ) + if output.exists(): + shutil.rmtree(output) + os.replace(temporary, output) + except BaseException: + shutil.rmtree(temporary, ignore_errors=True) + raise + return records + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("manifest", type=Path) + parser.add_argument("--output", type=Path, default=Path("build")) + parser.add_argument("--as-of", type=dt.date.fromisoformat) + args = parser.parse_args() + records = build(args.manifest, args.output, as_of=args.as_of) + print(f"{args.output}: published {len(records)} document(s)") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/check_currency.py b/tools/check_currency.py new file mode 100644 index 0000000..c7d1102 --- /dev/null +++ b/tools/check_currency.py @@ -0,0 +1,38 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import datetime as dt +from pathlib import Path + +from build_site import _add_interval, load_manifest +from render import split_frontmatter + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("manifest", type=Path) + parser.add_argument("--as-of", type=dt.date.fromisoformat, default=dt.date.today()) + args = parser.parse_args() + manifest_path = args.manifest.resolve() + manifest = load_manifest(manifest_path) + stale = 0 + for document in manifest["documents"]: + repo = manifest_path.parent / manifest["repositories"][document["source_repo"]]["path"] + source = (repo / document["source_path"]).resolve() + meta, _markdown = split_frontmatter(source.read_text(encoding="utf-8")) + reviewed = meta.get("last_reviewed") or meta.get("updated") + interval = document.get("review_interval") or meta.get("review_interval") + if not reviewed or not interval: + print(f"UNDECLARED {document['id']}: review date/interval missing") + stale += 1 + continue + due = _add_interval(reviewed, interval) + state = "STALE" if due < args.as_of else "current" + print(f"{state} {document['id']}: reviewed {reviewed}, due {due}") + stale += state == "STALE" + return 1 if stale else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/render.py b/tools/render.py index 9069bae..5d01653 100644 --- a/tools/render.py +++ b/tools/render.py @@ -19,7 +19,7 @@ Stdlib only, per the estate's structure-not-tooling stance: a publishing step that needs its own toolchain is a publishing step that stops being run. Usage: - python3 tools/render-artifact.py --output \\ + python3 tools/render.py --output \\ [--title "Name"] [--subtitle "..."] """ @@ -39,7 +39,7 @@ EM = re.compile(r"(? tuple[list[list[str]], int]: def render_ladder(rows: list[list[str]]) -> str: """A level table becomes a stepped scale. Colour depth encodes strength.""" body = rows[1:] - plane = LEVEL_CELL.match(body[0][0]).group(1) + axis = LEVEL_CELL.match(body[0][0]).group(1) names = { "I": "Identity", "A": "Authorization", "E": "Enforcement", - "P": "Placement", "R": "Retention", + "P": "Placement", "R": "Retention", "V": "Availability", } rungs = [] for cells in body: @@ -104,15 +104,15 @@ def render_ladder(rows: list[list[str]]) -> str: n = int(match.group(2)) text = cells[1] if len(cells) > 1 else "" rungs.append( - f'
{plane}{n}' + f'
{axis}{n}' f'{inline(text)}
' ) while len(rungs) < 5: rungs.append('
' - f'Ladder ends at {plane}{len(rungs) - 1}.
') + f'Ladder ends at {axis}{len(rungs) - 1}.
') return ( '
' - f'{names.get(plane, plane)}plane {plane}
' + f'{names.get(axis, axis)}axis {axis}
' f'
{"".join(rungs)}
' ) @@ -326,25 +326,30 @@ def render_body(markdown: str) -> tuple[str, list[tuple[str, str, str]]]: return "\n".join(out), rail -def main() -> int: - ap = argparse.ArgumentParser() - ap.add_argument("source", type=pathlib.Path) - ap.add_argument("--output", required=True, type=pathlib.Path) - ap.add_argument("--title", default=None) - ap.add_argument("--subtitle", default="") - args = ap.parse_args() - - meta, markdown = split_frontmatter(args.source.read_text()) +def render_page( + source: pathlib.Path, + *, + title: str | None = None, + subtitle: str = "", + publication: dict[str, str] | None = None, +) -> tuple[str, dict, int]: + meta, markdown = split_frontmatter(source.read_text(encoding="utf-8")) body, rail = render_body(markdown) - title = args.title or meta.get("title", args.source.stem) - display = title.split(":")[0].strip() + resolved_title = title or meta.get("title", source.stem) + display = resolved_title.split(":")[0].strip() eyebrow = " ".join( f"{html.escape(v)}" for k, v in ( ("id", meta.get("id", "")), ("status", f"{meta.get('status', '')} · {meta.get('revision', '')}".strip(" ·")), - ("date", meta.get("date", "")), + ("owner", meta.get("owner", "")), + ( + "review", + f"reviewed {meta.get('last_reviewed', '')}" + if meta.get("last_reviewed") + else "", + ), ) if v ) @@ -353,23 +358,103 @@ def main() -> int: for a, n, t in rail ) + publication = publication or {} + source_note = " · ".join( + item + for item in ( + publication.get("source_repo", ""), + publication.get("source_path", ""), + publication.get("source_revision", ""), + ) + if item + ) + source_line = ( + f'

Source: {html.escape(source_note)}

' + if source_note + else "" + ) + review_due = publication.get("review_due", "") + review_line = ( + f'

Review due: {html.escape(review_due)}

' + if review_due + else "" + ) + lifecycle = publication.get("lifecycle", "active") + successor = publication.get("successor", "") + lifecycle_notice = "" + if lifecycle == "superseded": + successor_link = ( + f' Read its successor.' + if successor + else "" + ) + lifecycle_notice = ( + '

Superseded.' + f" This address is retained as part of the policy record.{successor_link}

" + ) + elif lifecycle == "withdrawn": + lifecycle_notice = ( + '

Withdrawn. ' + "This document is retained for historical reference and is not current policy." + "

" + ) + currency_notice = ( + '

Review overdue. ' + f"This document was due for review on {html.escape(review_due)}.

" + if review_due and publication.get("stale") == "true" + else "" + ) + source_revision_meta = ( + f'\n' + if publication.get("source_revision") + else "" + ) + source_digest_meta = ( + f'\n' + if publication.get("source_digest") + else "" + ) page = ( - f"{html.escape(display)}\n" + "\n\n" + + source_revision_meta + + source_digest_meta + + f"{html.escape(display)}\n" f"\n" '
' - f'
{eyebrow}generated from canon — do not edit
' + f'
{eyebrow}generated from canonical source — do not edit
' f"

{html.escape(display)}

" - + (f'

{html.escape(args.subtitle)}

' if args.subtitle else "") + + (f'

{html.escape(subtitle)}

' if subtitle else "") + + source_line + + review_line + '
' f'' - f"
{body}" + f"
{lifecycle_notice}{currency_notice}{body}" f'
{html.escape(meta.get("id", ""))} · ' f'{html.escape(meta.get("revision", ""))} · {html.escape(meta.get("status", ""))}' - "generated from the-custodian/canon
" - "
\n" + f"{html.escape(source_note or 'generated from canonical source')}" + "\n" ) - args.output.write_text(page) - print(f"{args.output}: {len(rail)} sections, {len(page)} bytes") + return page, meta, len(rail) + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("source", type=pathlib.Path) + ap.add_argument("--output", required=True, type=pathlib.Path) + ap.add_argument("--title", default=None) + ap.add_argument("--subtitle", default="") + args = ap.parse_args() + + page, _meta, section_count = render_page( + args.source, + title=args.title, + subtitle=args.subtitle, + ) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(page, encoding="utf-8") + print(f"{args.output}: {section_count} sections, {len(page)} bytes") return 0 diff --git a/tools/verify_release.py b/tools/verify_release.py new file mode 100644 index 0000000..2750b8e --- /dev/null +++ b/tools/verify_release.py @@ -0,0 +1,126 @@ +#!/usr/bin/env python3 +"""Fail closed unless a built policy site is safe to publish as an OCI release.""" + +from __future__ import annotations + +import argparse +import hashlib +import html +import json +from pathlib import Path, PurePosixPath +import re +from typing import Any + + +HEX_DIGEST = re.compile(r"^[a-f0-9]{64}$") +CLEAN_GIT_REVISION = re.compile(r"^[a-f0-9]{40}$") +SOURCE_REVISION = re.compile( + r'' +) +SOURCE_DIGEST = re.compile(r'') + + +def _safe_relative(value: str) -> PurePosixPath: + path = PurePosixPath(value) + if path.is_absolute() or ".." in path.parts or not path.parts: + raise ValueError(f"unsafe release path: {value!r}") + return path + + +def _page_provenance(path: Path) -> tuple[str, str]: + page = path.read_text(encoding="utf-8") + revision = SOURCE_REVISION.search(page) + digest = SOURCE_DIGEST.search(page) + if not revision or not digest: + raise ValueError(f"{path}: missing source provenance metadata") + return html.unescape(revision.group(1)), html.unescape(digest.group(1)) + + +def verify(build: Path) -> dict[str, Any]: + build = build.resolve() + if not build.is_dir(): + raise ValueError(f"release directory does not exist: {build}") + for path in build.rglob("*"): + if path.is_symlink(): + raise ValueError(f"release tree contains a symlink: {path.relative_to(build)}") + + index = build / "index.html" + manifest_path = build / "publication-manifest.json" + if not index.is_file() or not manifest_path.is_file(): + raise ValueError("release requires index.html and publication-manifest.json") + + manifest_bytes = manifest_path.read_bytes() + manifest = json.loads(manifest_bytes) + if manifest.get("schema_version") != 1: + raise ValueError("publication manifest schema_version must be 1") + if not manifest.get("generated_as_of"): + raise ValueError("publication manifest generated_as_of is required") + documents = manifest.get("documents") + if not isinstance(documents, list) or not documents: + raise ValueError("publication manifest must contain at least one document") + + verified: list[str] = [] + for document in documents: + document_id = document.get("id", "") + for field in ( + "id", + "title", + "status", + "revision", + "owner", + "last_reviewed", + "review_due", + "canonical_path", + "revision_path", + ): + if not document.get(field) or document.get(field) == "unknown": + raise ValueError(f"{document_id}: release metadata field {field} is required") + source_revision = document.get("source_revision", "") + source_digest = document.get("source_digest", "") + if not CLEAN_GIT_REVISION.fullmatch(source_revision): + raise ValueError( + f"{document_id}: production source_revision must be a clean 40-hex Git commit; " + f"got {source_revision!r}" + ) + if not HEX_DIGEST.fullmatch(source_digest): + raise ValueError(f"{document_id}: invalid source_digest {source_digest!r}") + + canonical = build / _safe_relative(document["canonical_path"]) + revision = build / _safe_relative(document["revision_path"]) + if not canonical.is_file() or not revision.is_file(): + raise ValueError(f"{document_id}: canonical or immutable revision page is missing") + + current_revision, current_digest = _page_provenance(canonical) + immutable_revision, immutable_digest = _page_provenance(revision) + if (current_revision, current_digest) != (source_revision, source_digest): + raise ValueError(f"{document_id}: canonical page provenance differs from manifest") + if immutable_digest != source_digest: + raise ValueError(f"{document_id}: immutable revision digest differs from manifest") + if not CLEAN_GIT_REVISION.fullmatch(immutable_revision): + raise ValueError( + f"{document_id}: immutable revision page was not built from a clean Git commit" + ) + verified.append(document_id) + + return { + "schema_version": "policy-nexus-release/v1", + "publication_manifest_digest": hashlib.sha256(manifest_bytes).hexdigest(), + "generated_as_of": manifest["generated_as_of"], + "documents": verified, + } + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("build", nargs="?", type=Path, default=Path("build")) + args = parser.parse_args() + try: + evidence = verify(args.build) + except (KeyError, OSError, ValueError, json.JSONDecodeError) as exc: + parser.error(str(exc)) + print(json.dumps(evidence, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/workplans/POLICY-NEXUS-WP-0001-permanent-publication-surface.md b/workplans/POLICY-NEXUS-WP-0001-permanent-publication-surface.md index 776ab6a..3c71bdd 100644 --- a/workplans/POLICY-NEXUS-WP-0001-permanent-publication-surface.md +++ b/workplans/POLICY-NEXUS-WP-0001-permanent-publication-surface.md @@ -4,11 +4,11 @@ type: workplan title: "Stand up policy.coulomb.social as the permanent publication surface" domain: infotech repo: policy-nexus -status: proposed +status: active owner: the-custodian topic_slug: policy-nexus created: "2026-08-17" -updated: "2026-08-17" +updated: "2026-08-18" --- # POLICY-NEXUS-WP-0001 — permanent publication surface @@ -26,15 +26,15 @@ superseded versions, and can be reached by someone outside the estate. ## The forcing case `net-kingdom/canon/standards/tenancy-posture_v0.1.md` (*Tenancy Posture*, -draft-5) needs review from six repos — +draft-8) was reviewed by six repos — `tenant-engine`, `flex-auth`, `rapp-postgres`, `railiance-platform`, `adaptive-pricing`, `audit-core`. It is currently served from a private, disposable artifact URL. Routing a document that governs six repos to a link that may not resolve later is the problem this workplan exists to end. Tenancy Posture is therefore the first publication and the acceptance test. If the site -cannot carry it correctly — five ladders, a threat matrix, an E×P grid, twelve -owner-attributed open questions — the site is not finished. +cannot carry it correctly — six ladders, a threat matrix, an E×P grid, and its +owner-attributed questions and resolutions — the site is not finished. ## Existing structure this workplan must respect @@ -45,10 +45,11 @@ document rather than requiring extra markup. It was written because the page and the source had diverged. T02 generalises it from one document to many; it does not start from scratch. -**Source of truth stays upstream.** Canon lives in `the-custodian/canon`; -per-repo ADRs live in their own repos. This repo reads and never writes back. -A publication surface with write authority is a second source of truth, and the -estate has a standing rule against that. +**Source of truth stays upstream.** Canon lives in its owning repo — including +`the-custodian/canon` and `net-kingdom/canon` — while per-repo ADRs live in +their own repos. This repo reads and never writes back. A publication surface +with write authority is a second source of truth, and the estate has a standing +rule against that. **The adoption stance applies** (Tenancy Posture §14): adopt published standards and structural patterns; build tooling ground-up unless it is an established @@ -73,6 +74,12 @@ credential come from the existing platform. If this repo needs storage it is a ### T01 — Addressing scheme and permanence contract +```task +id: POLICY-NEXUS-WP-0001-T01 +status: done +priority: high +``` + Decide, once, how a document maps to a URL, and write down what the estate is promising about that URL. @@ -90,8 +97,18 @@ promising about that URL. **Why first:** everything downstream bakes in the answer, and changing it later breaks the one promise the repo exists to make. +Completed 2026-08-18 in `docs/adr/ADR-0001-addressing-and-permanence.md`. +The contract distinguishes stable current addresses from immutable revision +addresses and retains superseded and withdrawn publications. + ### T02 — Generalise the renderer +```task +id: POLICY-NEXUS-WP-0001-T02 +status: done +priority: high +``` + Lift `tools/render.py` from one document to many. - Multi-document: a manifest of sources rather than one path argument. @@ -100,13 +117,24 @@ Lift `tools/render.py` from one document to many. - Index generation: a landing page listing documents with status and currency. - Keep the convention recognisers (level ladders, threat matrix, E×P grid, evidence chips, section rail) and keep it stdlib-only. -- Keep the "generated from canon — do not edit" marker on every page. +- Keep the "generated from canonical source — do not edit" marker on every page. **Acceptance:** Tenancy Posture renders byte-identically in substance to the current generated page, plus an index entry. +Completed 2026-08-18. `publication.json` drives a stdlib-only multi-document +builder. It emits a current page, immutable revision, legacy alias, index and +machine-readable publication manifest. Tests cover path safety, revision +immutability, lifecycle notices and currency. + ### T03 — Source ingestion +```task +id: POLICY-NEXUS-WP-0001-T03 +status: progress +priority: high +``` + Define how a document gets from its owning repo to this one. - Manifest format: source repo, path, publication URL, owner. @@ -129,8 +157,19 @@ owning repo triggers on merge). Pull is simpler and keeps the direction of dependency clean; push is fresher. Recommend pull with a manual trigger, and record the choice. +2026-08-18: the explicit pull manifest and exact source-revision recording are +implemented, and missing or inconsistent sources fail the build. Tenancy +Posture is the first entry. Enumerating the remaining in-scope canon and ADR +corpus and connecting scheduled/manual checkout refresh remain open. + ### T04 — Deploy to policy.coulomb.social +```task +id: POLICY-NEXUS-WP-0001-T04 +status: progress +priority: high +``` + - DNS, TLS, ingress via the existing platform packages. - Static hosting — the output is static files by construction, so the serving layer should be the least interesting part of this workplan. @@ -141,8 +180,22 @@ record the choice. - Smoke check after every deploy: the index resolves, Tenancy Posture resolves, and a known superseded URL still resolves. +2026-08-18: `policy-nexus` now owns a digest-pinned, non-root OCI artifact +contract and a Forgejo publication workflow. `rapp-policy-nexus` owns the +separately reviewable Helm package, runtime hardening, public smoke checks and +rollback; `railiance-apps` owns only the production digest binding. DNS already +resolves and the reef has the required Traefik/cert-manager substrate. A clean +source release, registry digest, server-side dry-run, deployment and live smoke +evidence remain before this task can close. + ### T05 — Currency and staleness +```task +id: POLICY-NEXUS-WP-0001-T05 +status: done +priority: medium +``` + The relevance half of the repo's purpose. A permanently available document that is quietly out of date is worse than no document. @@ -154,8 +207,18 @@ is quietly out of date is worse than no document. unratified since 2026-08-10; that fact should be visible on the site, because invisibility is precisely why it stalled. +Completed 2026-08-18 for the published corpus. Pages and the index expose +review due dates and overdue state; `make currency` exits non-zero for stale or +undeclared review metadata. Expansion follows T03 automatically. + ### T06 — withdrawn +```task +id: POLICY-NEXUS-WP-0001-T06 +status: done +priority: low +``` + Regulatory intake moved to `risk-nexus` on 2026-08-17. Deciding what a rule demands of the estate is a judgement about risk, not an act of publishing, and it wanted a different owner and a different skill from everything above. @@ -218,11 +281,12 @@ The README's one-line description — "a convergence and publication point for government policies" — reads broader than this. Worth updating so the repo does not attract the wrong contributions. -## Open questions for the operator +## Resolved publication scope -1. **Canon subdirectory scope.** `standards` and `architecture` are clearly - policy. Are `constitution`, `values`, `tpsc` and `projects` in or out? T03 - needs a yes or no per directory rather than a wildcard. +ADR-0001 records the bounded answer: `standards`, `architecture` and +`constitution` are in; `values`, `tpsc` and `projects` are out unless their +owner explicitly identifies an individual governing document. T03 must still +enumerate each publication rather than globbing those trees. ## Deferred: controlled disclosure