diff --git a/assurance/service-contract.json b/assurance/service-contract.json index b2d8dba..2e3b06f 100644 --- a/assurance/service-contract.json +++ b/assurance/service-contract.json @@ -95,6 +95,10 @@ "forgejo-db.restore": { "owner": "railiance-platform", "max_age_seconds": 2592000 + }, + "eso.token-renewal": { + "owner": "railiance-platform", + "max_age_seconds": 129600 } } } diff --git a/docs/service-assurance.md b/docs/service-assurance.md index 5eb2731..d96aec0 100644 --- a/docs/service-assurance.md +++ b/docs/service-assurance.md @@ -54,6 +54,13 @@ lag guarantee. The fleet ESO aggregate is deliberately a conservative inventory check; an obsolete resource must be explicitly classified by its owner before it is excluded. One ESO failure cannot disappear inside an average. +`eso.token-renewal` (RPF-WP-0046) is the last successful run of the +`external-secrets/eso-token-renewer` CronJob. The CronJob keeps the periodic +parent tokens of the five dynamic-database stores alive. The signal reads +CronJob status timestamps only. It fails when a newer scheduled run has not +succeeded within an hour, and it goes stale after 36h. The tokens lapse after +7 days without renewal. + The following remain missing until a native value-safe adapter and acceptance exist: validated isolated restore receipts, OpenBao snapshot/restore proof, and offsite upload/restore receipts. Missing diff --git a/scripts/capture_service_observation.py b/scripts/capture_service_observation.py index 63aa25a..e17f0b0 100644 --- a/scripts/capture_service_observation.py +++ b/scripts/capture_service_observation.py @@ -36,6 +36,21 @@ def memory_bytes(value): return float(match.group(1)) * multipliers[unit] +def renewal_signal(status, now): + """Classify eso-token-renewer CronJob status (RPF-WP-0046) from timestamps only.""" + success = status.get('lastSuccessfulTime') or '' + scheduled = status.get('lastScheduleTime') or '' + if not success: + return 'unavailable', now.isoformat() + succeeded_at = datetime.fromisoformat(success.replace('Z', '+00:00')) + if scheduled: + scheduled_at = datetime.fromisoformat(scheduled.replace('Z', '+00:00')) + # A newer scheduled run that has not succeeded within an hour failed. + if (scheduled_at - succeeded_at).total_seconds() > 3600 and (now - scheduled_at).total_seconds() > 3600: + return 'fail', scheduled + return 'pass', success + + def capture(): contract = read_contract(ROOT / 'assurance/service-contract.json') uid = query(['get', 'namespace', 'kube-system', '-o', 'go-template={{printf "%q" .metadata.uid}}']) @@ -94,6 +109,13 @@ def capture(): except (ValueError, KeyError, TypeError, subprocess.TimeoutExpired): add('eso.ready', 'unavailable') add('eso.refresh', 'unavailable') + try: + status = query(['get', 'cronjob', 'eso-token-renewer', '-n', 'external-secrets', '-o', + 'go-template={"lastSuccessfulTime":{{printf "%q" (or .status.lastSuccessfulTime "")}},"lastScheduleTime":{{printf "%q" (or .status.lastScheduleTime "")}}}']) + result, when = renewal_signal(status, datetime.now(timezone.utc)) + add('eso.token-renewal', result, when) + except (ValueError, KeyError, TypeError, subprocess.TimeoutExpired): + add('eso.token-renewal', 'unavailable') # No token, Secret, application data/logs or seal/unseal mutation. # Receipt timestamps are preserved; reads never renew recovery evidence. signals.update(recovery_signals(datetime.now(timezone.utc))) diff --git a/tests/test_service_assurance.py b/tests/test_service_assurance.py index 01db559..2685c48 100644 --- a/tests/test_service_assurance.py +++ b/tests/test_service_assurance.py @@ -166,3 +166,21 @@ class CollectorTests(unittest.TestCase): if __name__ == '__main__': unittest.main() + + +def test_eso_token_renewal_signal_classification(): + import importlib.util + from datetime import datetime, timezone + from pathlib import Path + spec = importlib.util.spec_from_file_location( + 'capture', Path(__file__).resolve().parents[1] / 'scripts/capture_service_observation.py') + capture = importlib.util.module_from_spec(spec) + spec.loader.exec_module(capture) + now = datetime(2026, 9, 25, 12, 0, tzinfo=timezone.utc) + assert capture.renewal_signal({}, now)[0] == 'unavailable' + ok = {'lastSuccessfulTime': '2026-09-25T02:40:09Z', 'lastScheduleTime': '2026-09-25T02:40:00Z'} + assert capture.renewal_signal(ok, now) == ('pass', '2026-09-25T02:40:09Z') + failed = {'lastSuccessfulTime': '2026-09-24T02:40:09Z', 'lastScheduleTime': '2026-09-25T02:40:00Z'} + assert capture.renewal_signal(failed, now) == ('fail', '2026-09-25T02:40:00Z') + running = {'lastSuccessfulTime': '2026-09-24T02:40:09Z', 'lastScheduleTime': '2026-09-25T11:30:00Z'} + assert capture.renewal_signal(running, now)[0] == 'pass' diff --git a/workplans/RPF-WP-0046-eso-database-token-renewal.md b/workplans/RPF-WP-0046-eso-database-token-renewal.md index 4febfb6..fbf6a89 100644 --- a/workplans/RPF-WP-0046-eso-database-token-renewal.md +++ b/workplans/RPF-WP-0046-eso-database-token-renewal.md @@ -4,7 +4,7 @@ type: workplan title: "Keep the dynamic-database ESO parent tokens alive: periodic tokens and a renewer" domain: financials repo: railiance-platform -status: active +status: finished owner: railiance-platform topic_slug: railiance created: "2026-09-23" @@ -165,7 +165,7 @@ ready. ```task id: RPF-WP-0046-T06 -status: todo +status: done priority: medium state_hub_task_id: "2f528e56-ebf0-547e-8d4f-3cb34c348d46" ``` @@ -210,3 +210,10 @@ state_hub_task_id: "2f528e56-ebf0-547e-8d4f-3cb34c348d46" State Hub `/repos` returns 200. No workload now depends on a lease from the replaced max-TTL tokens. Those tokens expire on their own, from about 2026-10-10. +- **T06 done.** Hand-offs sent 2026-09-23. rapp-postgres and audit-core were + asked to retire or change their 768h mint scripts. railiance-telemetry was + asked for an alert on a failed run or no success for 48h. activity-core was + told the October re-mint is not needed. `assurance-capture` now emits + `eso.token-renewal` (CronJob status timestamps only, 36h budget), and it is + healthy on the first live check. The owners' follow-through is tracked in + their own repos.