Advance blocked assurance and operator callback work
All checks were successful
CI Smoke / host-smoke (push) Successful in 0s
CI Smoke / container-smoke (push) Successful in 1s

Assistant: codex
Assistant-Model: gpt-6-astra
Assistant-Session: 01a06ecb-456a-71c2-b41e-0755d336e883
This commit is contained in:
codex 2026-09-06 14:16:49 +02:00
parent a3ca4b708f
commit 445f1361dc
16 changed files with 505 additions and 144 deletions

View file

@ -0,0 +1,15 @@
{
"schema": "railiance-platform.recovery-evidence.v1",
"receipts": [
{
"signal": "apps-pg.restore",
"path": "docs/evidence/scaleway-primary-restore-2026-09-06.json",
"sha256": "c9f7c64011f909033f9e8c074adbad0831ec3ef616997225163d34b075b35739"
},
{
"signal": "forgejo-db.restore",
"path": "docs/evidence/forgejo-scaleway-restore-2026-09-06.json",
"sha256": "071a732318a55a8ca152d90b7a4cd70644728f94ce2753ab73d1c94ba2a26114"
}
]
}

View file

@ -91,6 +91,10 @@
"eso.refresh": {
"owner": "railiance-platform",
"max_age_seconds": 3600
},
"forgejo-db.restore": {
"owner": "railiance-platform",
"max_age_seconds": 2592000
}
}
}

View file

@ -140,7 +140,7 @@
"consumers": [
"forgejo"
],
"failure_domain": "source host plus separately encrypted Nextcloud copy",
"failure_domain": "source host, Scaleway full archive and native database backup, independent encrypted Nextcloud essentials copy",
"availability": {
"status": "unsupported",
"target": null,
@ -156,12 +156,12 @@
"target_seconds": null,
"decision_owner": "railiance-platform + railiance-forge"
},
"retention": "Nextcloud secondary account: 10 GiB total; 14 daily + 4 weekly is an unfulfilled target at the measured 5.35 GB/archive size",
"retention": "Scaleway native database: 30 days. Nextcloud essentials: proposed 7 daily + 2 weekly within 10 GiB; scheduled tier/retention cutover remains pending in RPF-WP-0038-T04.",
"existing_evidence": "docs/forgejo-backup.md",
"recovery_custody": "OpenBao: 2-of-3 operator quorum plus independent encrypted snapshot custody; database/offsite: governed backup lane and separately available restore key. Availability not verified in this task.",
"maintenance_abort": "docs/railiance01-coordinated-reboot.md; stop before mutation when freshness, quorum, consumer readiness or named abort operator is absent",
"freshness_policy": "assurance/service-contract.json; diagnostic thresholds only, no installed cadence approval",
"requirement_assessment": "No accepted numeric consumer availability/RPO/RTO requirement found in the reviewed contracts. Service classes inform placement, not guarantees. Refuse any request for guaranteed HA/node-loss recovery until matched to supported substrate and package proof. Scaleway is the selected primary provider, but no Forgejo primary archive or native database destination is evidenced; coverage remains incomplete."
"requirement_assessment": "Scaleway native Forgejo database recovery and full application archive recovery passed on 2026-09-06; independent Nextcloud essentials recovery passed with packages disabled. See RPF-WP-0038 evidence. No accepted numeric availability/RPO/RTO guarantees or recurring tiered archive cadence are established."
},
{
"service": "cnpg-option-a",

View file

@ -0,0 +1,180 @@
{
"observation": {
"schema": "railiance-platform.observation.v1",
"cluster_uid": "a553c742-0115-43d4-99a4-a5ca56fe0786",
"captured_at": "2026-09-06T10:13:09.536959+00:00",
"signals": {
"apps-pg.ready": {
"result": "pass",
"observed_at": "2026-09-06T10:13:01.557274+00:00"
},
"apps-pg.backup": {
"result": "pass",
"observed_at": "2026-09-06T02:15:09Z"
},
"apps-pg.wal": {
"result": "pass",
"observed_at": "2026-09-06T10:13:01.557317+00:00"
},
"apps-pg.headroom": {
"result": "pass",
"observed_at": "2026-09-06T10:12:43Z"
},
"platform-pg.ready": {
"result": "pass",
"observed_at": "2026-09-06T10:13:03.817521+00:00"
},
"platform-pg.backup": {
"result": "pass",
"observed_at": "2026-09-06T02:15:15Z"
},
"platform-pg.wal": {
"result": "pass",
"observed_at": "2026-09-06T10:13:03.817548+00:00"
},
"platform-pg.headroom": {
"result": "pass",
"observed_at": "2026-09-06T10:12:59Z"
},
"platform-pg-2.ready": {
"result": "pass",
"observed_at": "2026-09-06T10:13:06.269440+00:00"
},
"platform-pg-2.backup": {
"result": "pass",
"observed_at": "2026-09-06T02:15:09Z"
},
"platform-pg-2.wal": {
"result": "pass",
"observed_at": "2026-09-06T10:13:06.269464+00:00"
},
"platform-pg-2.headroom": {
"result": "pass",
"observed_at": "2026-09-06T10:12:56Z"
},
"openbao.seal": {
"result": "pass",
"observed_at": "2026-09-06T10:13:08.833110+00:00"
},
"eso.ready": {
"result": "pass",
"observed_at": "2026-09-06T10:13:09.535776+00:00"
},
"eso.refresh": {
"result": "pass",
"observed_at": "2026-09-06T09:13:29Z"
},
"apps-pg.restore": {
"result": "pass",
"observed_at": "2026-09-05T22:30:45.208512+00:00"
},
"forgejo-db.restore": {
"result": "pass",
"observed_at": "2026-09-05T22:55:54.893587+00:00"
}
}
},
"evaluation": {
"schema": "railiance-platform.assurance-signal.v1",
"cluster_uid": "a553c742-0115-43d4-99a4-a5ca56fe0786",
"evaluated_at": "2026-09-06T10:18:41.042228+00:00",
"signals": {
"apps-pg.ready": {
"state": "healthy",
"owner": "railiance-platform"
},
"apps-pg.backup": {
"state": "healthy",
"owner": "railiance-platform"
},
"apps-pg.wal": {
"state": "healthy",
"owner": "railiance-platform"
},
"apps-pg.restore": {
"state": "healthy",
"owner": "railiance-platform"
},
"apps-pg.headroom": {
"state": "healthy",
"owner": "railiance-platform"
},
"platform-pg.ready": {
"state": "healthy",
"owner": "rapp-postgres"
},
"platform-pg.backup": {
"state": "healthy",
"owner": "rapp-postgres"
},
"platform-pg.wal": {
"state": "healthy",
"owner": "rapp-postgres"
},
"platform-pg.restore": {
"state": "missing",
"owner": "rapp-postgres"
},
"platform-pg.headroom": {
"state": "healthy",
"owner": "rapp-postgres"
},
"platform-pg-2.ready": {
"state": "healthy",
"owner": "rapp-postgres"
},
"platform-pg-2.backup": {
"state": "healthy",
"owner": "rapp-postgres"
},
"platform-pg-2.wal": {
"state": "healthy",
"owner": "rapp-postgres"
},
"platform-pg-2.restore": {
"state": "missing",
"owner": "rapp-postgres"
},
"platform-pg-2.headroom": {
"state": "healthy",
"owner": "rapp-postgres"
},
"openbao.seal": {
"state": "healthy",
"owner": "railiance-platform"
},
"openbao.snapshot": {
"state": "missing",
"owner": "railiance-platform"
},
"openbao.restore": {
"state": "missing",
"owner": "railiance-platform"
},
"offsite.upload": {
"state": "missing",
"owner": "railiance-platform"
},
"offsite.restore": {
"state": "missing",
"owner": "railiance-platform"
},
"eso.ready": {
"state": "healthy",
"owner": "railiance-platform"
},
"eso.refresh": {
"state": "stale",
"owner": "railiance-platform"
},
"forgejo-db.restore": {
"state": "healthy",
"owner": "railiance-platform"
}
},
"transport": "unmonitored",
"guarantees": "unsupported",
"threshold_status": "local-diagnostic-only",
"healthy": false
}
}

View file

@ -103,3 +103,15 @@ Source admission checks happen before deployment in the existing apps-pg and
package owner paths; this additional check detects disclosure drift. It does
not apply resources or become a second provisioning engine. Any new consumer
needs an owner entry and a reviewed baseline update, even within free capacity.
## Native recovery receipt adapter
`capture_service_observation.py` loads `assurance/recovery-evidence.json` through
`recovery_evidence.py`. The reviewed SHA-256 pins bind apps-pg and forgejo-db
restore samples to their Scaleway receipts. Original `finished_at` values drive
freshness; recapturing cannot extend their 30-day diagnostic validity. Missing,
changed, invalid or incorrectly scoped receipts yield unavailable samples.
Update pins only after reviewing replacement evidence. Native database recovery
does not attest full application or essentials recovery. Legacy archive receipts
without completion timestamps remain manual evidence; no timestamp is inferred
from file modification time. Automatic cadence and alert delivery remain pending.

View file

@ -0,0 +1,42 @@
# Blocked workplan progress — 2026-09-06
Prior work was clean at a3ca4b708f224de23292c30966b7fa1e52957cd5. Repo-manager
confirmed origin/main synchronized and primary railiance01 applied that exact
commit. Reviewed the six blocked plans against current source and recovery
receipts; no whole-plan closure gate is fully satisfied.
- **WP-0036 / T03:** implemented hash-pinned native Scaleway recovery adapters
for apps-pg and forgejo-db, preserving actual completion times. Invalid,
changed, future, incomplete-cleanup and wrong-provider receipts cannot pass.
Corrected stale Forgejo coverage records. Fresh capture/evaluation is persisted
in `docs/evidence/RPF-WP-0036-assurance-2026-09-06.json`: 16 healthy signals,
six missing recovery signals, one stale ESO refresh aggregate. Capture's
oldest refresh was 09:13:29Z, near the one-hour diagnostic boundary; subsequent
read-only inspection found all 27 ExternalSecrets Ready with newer refreshes.
This is not proof of an ESO outage. Accepted cadence/grace and Q2 failure/
absence delivery still need resolution. Archive receipts without completion
timestamps remain manual evidence, and essentials never attest full recovery.
- **WP-0025 / T03:** replaced whole-role hardcoded callback writes with a silent
preserving update, idempotence, observed-drift refusal and full readback checks.
Existing settings and callbacks survive. No CAS exists, so concurrent role
administration remains a limitation. No live role update was attempted;
attended loopback UI login still gates listener retraction.
- **WP-0035:** refreshed dependency assessment: secrets-engine implemented its
local claim/validation join; external approval-engine/access-engine serving
endpoints remain unavailable in its September 6 source. Service issuer and
operator group/tenant acceptance remain necessary. Signing T04 is complete.
- **WP-0027:** incident closure still needs predecessor disposition and canonical
custody acceptance. Do not reconstruct the unavailable predecessor secret.
- **WP-0029:** new Backup account recovery is proven; old Bernd share invalidation
evidence is still missing. Repeating backup drills would not close this gate.
- **WP-0015:** disruptive load/reboot exercises need fresh execution windows,
named abort operator and recovery prerequisites; earlier NO-GO is terminal.
WP-0038 remains active: scheduled primary/full and Nextcloud/essentials cutover
and retention execution are not established by one-off recovery proofs. Source
ownership and stale hub alias acceptance remain outstanding in WP-0036-T06.
No external owner acceptance or message delivery was asserted.
Validation: 28 focused tests passed (assurance evaluator/collector, recovery
receipt rejection/expiry, callback preservation/drift/silence). Live capture
completed, and a read-only ExternalSecret metadata query confirmed 27 Ready.

View file

@ -14,6 +14,7 @@ import sys
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / 'scripts'))
from service_assurance import read_contract, admission
from recovery_evidence import recovery_signals
def query(args, allowed_codes=(0,)):
@ -94,7 +95,8 @@ def capture():
add('eso.ready', 'unavailable')
add('eso.refresh', 'unavailable')
# No token, Secret, application data/logs or seal/unseal mutation.
# Native restore and offsite receipts remain separate attended evidence.
# Receipt timestamps are preserved; reads never renew recovery evidence.
signals.update(recovery_signals(datetime.now(timezone.utc)))
return {'schema': 'railiance-platform.observation.v1', 'cluster_uid': uid,
'captured_at': datetime.now(timezone.utc).isoformat(), 'signals': signals}

View file

@ -1,63 +1,4 @@
#!/usr/bin/env bash
# Silent owner command for the governed openbao-platform-admin-login lane.
# Warden rejects any child output and self-revokes the attended session after
# this command exits. The role payload contains no secret values.
# Silent child for the governed attended login; no role or token output.
set -euo pipefail
ROLE_PATH="auth/netkingdom/role/platform-admin"
CALLBACK_URI="http://127.0.0.1:18200/ui/vault/auth/netkingdom/oidc/callback"
PAYLOAD="$(mktemp "${TMPDIR:-/tmp}/openbao-platform-admin-role.XXXXXX.json")"
READBACK="$(mktemp "${TMPDIR:-/tmp}/openbao-platform-admin-readback.XXXXXX.json")"
cleanup() {
rm -f "$PAYLOAD" "$READBACK"
}
trap cleanup EXIT INT TERM
chmod 0600 "$PAYLOAD" "$READBACK"
command -v bao >/dev/null 2>&1
command -v python3 >/dev/null 2>&1
cat >"$PAYLOAD" <<'ROLE_JSON'
{
"role_type": "oidc",
"user_claim": "sub",
"groups_claim": "groups",
"oidc_scopes": ["openid", "profile", "email", "groups"],
"allowed_redirect_uris": [
"http://localhost:8250/oidc/callback",
"http://127.0.0.1:8250/oidc/callback",
"http://127.0.0.1:18200/ui/vault/auth/netkingdom/oidc/callback",
"https://bao.coulomb.social/ui/vault/auth/netkingdom/oidc/callback",
"https://bao.coulomb.social/ui/vault/auth/keycape/oidc/callback"
],
"bound_claims": {
"groups": ["net-kingdom-admins"]
},
"claim_mappings": {
"email": "email",
"preferred_username": "username"
},
"policies": ["platform-admin"],
"ttl": "1h"
}
ROLE_JSON
bao write "$ROLE_PATH" @"$PAYLOAD" >/dev/null 2>&1
bao read -format=json "$ROLE_PATH" >"$READBACK" 2>/dev/null
python3 - "$READBACK" "$CALLBACK_URI" <<'PY' >/dev/null 2>&1
import json
import sys
path, callback = sys.argv[1:]
with open(path, encoding="utf-8") as handle:
role = json.load(handle).get("data") or {}
if callback not in role.get("allowed_redirect_uris", []):
raise SystemExit(1)
if role.get("role_type") != "oidc":
raise SystemExit(1)
if "platform-admin" not in role.get("token_policies", role.get("policies", [])):
raise SystemExit(1)
PY
exec python3 "$(dirname "$0")/openbao_operator_loopback_callback.py" "$@" >/dev/null 2>&1

View file

@ -0,0 +1,49 @@
#!/usr/bin/env python3
"""Silent contained callback update; preserve the existing administrator role."""
import json
import subprocess
import sys
ROLE = 'auth/netkingdom/role/platform-admin'
CALLBACK = 'http://127.0.0.1:18200/ui/vault/auth/netkingdom/oidc/callback'
def read_role():
result = subprocess.run(['bao', 'read', '-format=json', ROLE],
capture_output=True, check=True, timeout=30)
role = json.loads(result.stdout)['data']
if (role.get('role_type') != 'oidc'
or 'platform-admin' not in role.get('token_policies', role.get('policies', []))
or not isinstance(role.get('allowed_redirect_uris'), list)
or not all(isinstance(uri, str) for uri in role['allowed_redirect_uris'])):
raise ValueError('unexpected role')
return role
def update(read=read_role, write=None):
original = read()
if CALLBACK in original['allowed_redirect_uris']:
return False
desired = dict(original, allowed_redirect_uris=original['allowed_redirect_uris'] + [CALLBACK])
if read() != original:
raise ValueError('role changed before write')
# The endpoint has no CAS: this detects observed drift, not an atomic lock.
if write is None:
subprocess.run(['bao', 'write', ROLE, '-'], input=json.dumps(desired).encode(),
capture_output=True, check=True, timeout=30)
else:
write(desired)
if read() != desired:
raise ValueError('role readback differs')
return True
if __name__ == '__main__':
try:
if sys.argv[1:] == ['--check-only']:
sys.exit(0 if CALLBACK in read_role()['allowed_redirect_uris'] else 3)
if sys.argv[1:]:
sys.exit(2)
update()
except Exception:
sys.exit(1)

View file

@ -0,0 +1,44 @@
"""Hash-pinned native recovery receipts, with original completion timestamps."""
import hashlib
import json
from pathlib import Path
from service_assurance import timestamp
ROOT = Path(__file__).resolve().parents[1]
def recovery_signals(now, root=ROOT):
index = json.loads((root / 'assurance/recovery-evidence.json').read_text())
if index['schema'] != 'railiance-platform.recovery-evidence.v1':
raise ValueError('unknown recovery index')
signals = {}
for entry in index['receipts']:
signal = entry['signal']
if signal in signals or signal not in ('apps-pg.restore', 'forgejo-db.restore'):
raise ValueError('unexpected recovery signal')
sample = {'result': 'unavailable', 'observed_at': now.isoformat()}
try:
path = (root / entry['path']).resolve()
if not path.is_relative_to((root / 'docs/evidence').resolve()):
raise ValueError('receipt outside evidence directory')
raw = path.read_bytes()
if hashlib.sha256(raw).hexdigest() != entry['sha256']:
raise ValueError('receipt drift')
receipt = json.loads(raw)
cell = signal.removesuffix('.restore')
if (receipt['schema'] != 'platform.scaleway-primary-restore.v1'
or receipt['primary_destination'] != f's3://railiance-platform-pg-backup/platform-pg/{cell}/'
or receipt['source'] != 'Scaleway Barman base backup and WAL'
or receipt['stage'] != 'database_acceptance'
or receipt['status'] != 'verified'
or receipt['cleanup'] is not True
or receipt['production_ready'] is not True):
raise ValueError('receipt not accepted')
completed = timestamp(receipt['finished_at'])
if not timestamp(receipt['started_at']) <= completed <= now:
raise ValueError('invalid receipt chronology')
sample = {'result': 'pass', 'observed_at': receipt['finished_at']}
except (OSError, ValueError, KeyError, TypeError):
pass
signals[signal] = sample
return signals

View file

@ -1,78 +1,62 @@
import json
import os
import subprocess
import copy
import importlib.util
from pathlib import Path
import pytest
spec = importlib.util.spec_from_file_location('callback', Path(__file__).resolve().parents[1] / 'scripts/openbao_operator_loopback_callback.py')
m = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m)
REPO_ROOT = Path(__file__).resolve().parents[1]
SCRIPT = REPO_ROOT / "scripts/openbao-apply-operator-loopback-callback.sh"
CALLBACK = "http://127.0.0.1:18200/ui/vault/auth/netkingdom/oidc/callback"
def role():
return {'role_type': 'oidc', 'token_policies': ['platform-admin'],
'allowed_redirect_uris': ['https://existing.example/callback'],
'token_ttl': 900, 'bound_claims': {'groups': ['custom-admins']},
'claim_mappings': {'email': 'email'}, 'token_max_ttl': 1800}
def _fake_bao(tmp_path: Path) -> tuple[Path, Path]:
capture = tmp_path / "written-role.json"
executable = tmp_path / "bao"
executable.write_text(
"""#!/bin/sh
set -eu
if [ "$1" = "write" ]; then
cp "${3#@}" "$BAO_CAPTURE"
exit 0
fi
if [ "$1" = "read" ]; then
if [ "${BAO_FAKE_MISSING_CALLBACK:-false}" = "true" ]; then
printf '%s\\n' '{"data":{"role_type":"oidc","token_policies":["platform-admin"],"allowed_redirect_uris":[]}}'
else
printf '%s\\n' '{"data":{"role_type":"oidc","token_policies":["platform-admin"],"allowed_redirect_uris":["http://127.0.0.1:18200/ui/vault/auth/netkingdom/oidc/callback"]}}'
fi
exit 0
fi
exit 2
""",
encoding="utf-8",
)
def test_preserves_all_other_settings():
current = role()
original = copy.deepcopy(current)
writes = []
def write(value):
writes.append(value)
current.update(value)
assert m.update(lambda: copy.deepcopy(current), write)
assert len(writes) == 1
assert current == dict(original, allowed_redirect_uris=original['allowed_redirect_uris'] + [m.CALLBACK])
assert not m.update(lambda: copy.deepcopy(current), write)
assert len(writes) == 1
def test_observed_concurrent_change_prevents_write():
reads = iter([role(), dict(role(), token_ttl=300)])
with pytest.raises(ValueError):
m.update(lambda: next(reads), lambda _: pytest.fail('must not write'))
def test_failed_readback_is_not_success():
with pytest.raises(ValueError):
m.update(role, lambda _: None)
@pytest.mark.parametrize('mutation', [{'role_type': 'jwt'}, {'token_policies': ['other']}, {'allowed_redirect_uris': 'bad'}])
def test_unexpected_live_role_refused(monkeypatch, mutation):
import json
import subprocess
payload = dict(role(), **mutation)
monkeypatch.setattr(m.subprocess, 'run', lambda *a, **kw: subprocess.CompletedProcess(a, 0, json.dumps({'data': payload}).encode()))
with pytest.raises(ValueError):
m.read_role()
def test_silent_entrypoint_on_command_failure(tmp_path):
import os
import subprocess
executable = tmp_path / 'bao'
executable.write_text('#!/bin/sh\necho SECRET_CANARY >&2\nexit 1\n')
executable.chmod(0o755)
return executable, capture
def _run(tmp_path: Path, *, missing_callback: bool = False) -> tuple[subprocess.CompletedProcess[str], Path]:
_, capture = _fake_bao(tmp_path)
env = dict(os.environ)
env.update(
{
"PATH": f"{tmp_path}:{env['PATH']}",
"TMPDIR": str(tmp_path),
"BAO_CAPTURE": str(capture),
"BAO_FAKE_MISSING_CALLBACK": str(missing_callback).lower(),
}
)
result = subprocess.run(
[str(SCRIPT)],
cwd=REPO_ROOT,
env=env,
text=True,
capture_output=True,
check=False,
)
return result, capture
def test_contained_command_is_silent_and_writes_exact_role(tmp_path: Path) -> None:
result, capture = _run(tmp_path)
assert result.returncode == 0
assert result.stdout == ""
assert result.stderr == ""
role = json.loads(capture.read_text(encoding="utf-8"))
assert CALLBACK in role["allowed_redirect_uris"]
assert role["policies"] == ["platform-admin"]
assert role["bound_claims"] == {"groups": ["net-kingdom-admins"]}
assert not list(tmp_path.glob("openbao-platform-admin-*.json"))
def test_verification_failure_remains_silent_and_nonzero(tmp_path: Path) -> None:
result, _ = _run(tmp_path, missing_callback=True)
result = subprocess.run([str(Path(m.__file__).with_name('openbao-apply-operator-loopback-callback.sh'))],
env=dict(os.environ, PATH=str(tmp_path) + ':' + os.environ['PATH']), capture_output=True)
assert result.returncode != 0
assert result.stdout == ""
assert result.stderr == ""
assert not list(tmp_path.glob("openbao-platform-admin-*.json"))
assert result.stdout == result.stderr == b''

View file

@ -0,0 +1,43 @@
from datetime import datetime, timezone, timedelta
import hashlib
import json
from pathlib import Path
import sys
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'scripts'))
from recovery_evidence import recovery_signals, ROOT
from service_assurance import evaluate
NOW = datetime(2026, 9, 6, 12, tzinfo=timezone.utc)
def test_real_receipts_preserve_completion_and_eventually_expire():
signals = recovery_signals(NOW)
assert all(s['result'] == 'pass' for s in signals.values())
assert signals['apps-pg.restore']['observed_at'] == '2026-09-05T22:30:45.208512+00:00'
contract = {'cluster_uid': 'test', 'capture_max_age_seconds': 900,
'signals': {key: {'owner': 'platform', 'max_age_seconds': 2592000} for key in signals}}
later = NOW + timedelta(days=31)
result = evaluate(contract, {'schema': 'railiance-platform.observation.v1',
'cluster_uid': 'test', 'captured_at': later.isoformat(), 'signals': recovery_signals(later)}, later)
assert all(s['state'] == 'stale' for s in result['signals'].values())
@pytest.mark.parametrize('change', ['hash', 'cleanup', 'provider', 'missing_time', 'future', 'naive'])
def test_invalid_receipt_is_unavailable(tmp_path, change):
index = json.loads((ROOT / 'assurance/recovery-evidence.json').read_text())
index['receipts'] = index['receipts'][:1]
entry = index['receipts'][0]
receipt = json.loads((ROOT / entry['path']).read_text())
if change == 'cleanup': receipt['cleanup'] = False
if change == 'provider': receipt['primary_destination'] = 's3://other/'
if change == 'missing_time': del receipt['finished_at']
if change == 'future': receipt['finished_at'] = '2027-01-01T00:00:00Z'
if change == 'naive': receipt['finished_at'] = '2026-09-05T23:00:00'
path = tmp_path / entry['path']
path.parent.mkdir(parents=True)
path.write_text(json.dumps(receipt))
if change != 'hash': entry['sha256'] = hashlib.sha256(path.read_bytes()).hexdigest()
(tmp_path / 'assurance').mkdir()
(tmp_path / 'assurance/recovery-evidence.json').write_text(json.dumps(index))
assert recovery_signals(NOW, tmp_path)['apps-pg.restore']['result'] == 'unavailable'

View file

@ -147,7 +147,11 @@ class CollectorTests(unittest.TestCase):
baseline = json.loads((ROOT / 'assurance/admission-baseline.json').read_text())
with patch.object(self.collector, 'query', side_effect=query), patch.object(self.collector, 'admission', return_value=baseline):
observation = self.collector.capture()
for sample in observation['signals'].values(): self.assertEqual(sample['result'], 'unavailable')
for name, sample in observation['signals'].items():
if name not in ('apps-pg.restore', 'forgejo-db.restore'):
self.assertEqual(sample['result'], 'unavailable')
# Recorded recovery evidence is independent of failed live status reads.
self.assertEqual(observation['signals']['apps-pg.restore']['result'], 'pass')
self.assertNotIn('secret', [str(a).lower() for call in calls for a in call])
result = m.evaluate(contract, observation, datetime.now(timezone.utc))
self.assertFalse(result['healthy'])

View file

@ -8,7 +8,7 @@ status: blocked
owner: codex
topic_slug: railiance
created: "2026-08-23"
updated: "2026-09-05"
updated: "2026-09-06"
related:
- RMASTER-WP-0020-T09
- RAPP-OPENBAO-WP-0002
@ -96,3 +96,14 @@ issuer callback, ops-bridge the tunnel, and S1/S2 DNS/network primitives.
Unblock with a fresh attended OIDC/MFA callback update and loopback login,
then the guarded retraction and owner-specific DNS handoff. Existing source
readiness is not evidence of a completed live cutover.
## Callback preservation repair — 2026-09-06
T03 advanced locally: the attended callback helper now reads and preserves the
existing platform-admin role, appends only the exact loopback callback, skips
writes when already present, detects observed drift before writing, and verifies
all settings on readback. `--check-only` is silent and returns 3 if absent.
The role endpoint has no CAS; exclusive attended administration is still needed.
Tests cover settings preservation, idempotence, drift, readback failure and
unexpected roles. No live role update or ingress retraction was performed in
this follow-up; attended loopback UI login remains the cutover gate.

View file

@ -7,7 +7,7 @@ repo: railiance-platform
status: blocked
owner: codex
created: "2026-09-05"
updated: "2026-09-05"
updated: "2026-09-06"
related:
- RPF-WP-0032
- RPF-WP-0033
@ -140,3 +140,12 @@ signature, with healthy primary identity and no preflight blockers. State Hub
chart commit `49e3182`, Helm revision 59. No repository rename executed.
Evidence: `docs/evidence/RPF-WP-0035-T04-signing-activation-2026-09-05.json`;
closure: `history/2026-09-05-preflight-signing-activation-complete.md`.
## Dependency review — 2026-09-06
SECRETS-WP-0008-T02 now records the local PIP claim/validation join implemented
and tested. Its remaining gate is the unreachable approval-engine claim endpoint
and access-engine Check (SECRETS-WP-0007-T04). Do not carry forward the old local
stub as a blocker. T02 still needs the accepted service issuer/JWKS, claims,
audience and consumer binding. T03 still needs confirmed operator group/tenant
and consumer semantics; T04 is already complete. No new owner acceptance inferred.

View file

@ -7,7 +7,7 @@ repo: railiance-platform
status: blocked
owner: codex
created: "2026-09-05"
updated: "2026-09-05"
updated: "2026-09-06"
state_hub_workstream_id: "ca639c3d-3a87-5fa4-ad13-6f2e014b0c84"
---
@ -249,11 +249,32 @@ Scaleway is the selected primary; Nextcloud is the independent secondary.
Fresh apps-pg recovery from Scaleway passed in 42.64 seconds with expected
consumer databases and limits, production Ready and scratch cleanup complete.
Evidence: `docs/evidence/scaleway-primary-restore-2026-09-06.json`.
T03 now has this fresh physical recovery receipt but still lacks recurring
cadence, the other recovery surfaces and validated evidence adapters.
T03 now has this fresh physical recovery receipt; recurring assurance cadence
and the remaining recovery surfaces are still incomplete.
The source/live coverage inventory `docs/backup-provider-coverage.md` exposes
missing native primary configuration on forgejo-db/net-kingdom-pg/state-hub-db
and no reviewed Scaleway Forgejo archive destination. Track primary coverage
remaining native primary configuration gaps on net-kingdom-pg/state-hub-db.
Forgejo native database recovery and full Scaleway archive recovery have since
passed; independent Nextcloud essentials recovery also passed (WP-0038). Track primary coverage
here with forge/package/storage owners; do not silently claim the Nextcloud
account cutover filled it or weaken WP-0029's separate incident closure.
## Recovery evidence adapter follow-up — 2026-09-06
T03 advanced: capture now consumes hash-pinned apps-pg and forgejo-db native
Scaleway restore receipts through `scripts/recovery_evidence.py`. Completion
timestamps survive every capture; wrong destination, drift, missing/naive/future
times and unsuccessful cleanup cannot become passing samples. Forgejo database
restore is its own signal. Older application archive receipts lack completion
timestamps and remain outside the automatic adapter; essentials evidence never
substitutes for full recovery. Six previous recovery signals still lack adapters
or accepted evidence. Cadence, independent custody and Q2 transport gates remain.
The Forgejo service record now reflects the completed native/full/essentials
proofs while retaining unsupported guarantees and the pending scheduled cutover.
Live adapter verification: `docs/evidence/RPF-WP-0036-assurance-2026-09-06.json`
reports 16 healthy, six missing and one stale ESO refresh signal. The aggregate
crossed the one-hour diagnostic boundary; subsequent metadata inspection found
all 27 ExternalSecrets Ready with newer refreshes. Cadence/grace acceptance is
still needed; no outage or successful alert transport is inferred.