From bdbef582599a59a1f8416ac56ca1667d849e9e46 Mon Sep 17 00:00:00 2001 From: tegwick Date: Tue, 8 Sep 2026 10:48:35 +0200 Subject: [PATCH] Lease-watch verdict must check what it claims MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Version 3 reported 'survived past the lease TTL' after a clean 12-minute window on a 13-minute-old pod. Every measurement in it was accurate; the conclusion was not, because nothing checked that the observation window had actually exceeded the 30-minute lease it claimed to have outlasted. Fifth defect in this instrument, and the first to err toward reassurance. Versions 1 and 2 cried wolf, which provokes investigation. This one would have been believed, and readiness_state: verified recorded on it — the same way live-image-digest-match would have been believed. A check reporting success it has not established is indistinguishable from one that works, until it matters. The verdict now requires uptime > lease TTL and reports INCONCLUSIVE when a window is clean but too short. 'Clean' and 'proven' are different claims and only one of them was being measured. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01Bjefh8NUiEiahN4JLwoSKM Assistant: claude-code Assistant-Model: opus Assistant-Process: 388925@bnt-lap001 Assistant-Session: 3507023f-e0fd-4a1e-9d90-a0d4217d1502 --- WORK-RECORDS.md | 2 +- declarations/rapp.yaml | 6 +++--- ...WP-0002-T04-first-deployment-2026-09-08.md | 17 +++++++++++++++ ...-WP-0002-T05-lease-rotation-2026-09-08.log | 2 ++ manifests/migration.yaml | 4 ++-- manifests/runtime.yaml | 2 +- tools/lease-watch.sh | 21 +++++++++++++++---- tools/smoke.sh | 13 +++++++++++- 8 files changed, 55 insertions(+), 12 deletions(-) create mode 100644 docs/evidence/RCP-WP-0002-T05-lease-rotation-2026-09-08.log diff --git a/WORK-RECORDS.md b/WORK-RECORDS.md index b78789d..9b7eee4 100644 --- a/WORK-RECORDS.md +++ b/WORK-RECORDS.md @@ -16,5 +16,5 @@ | task | RCP-WP-0002-T01 | done | — | workplans/RCP-WP-0002-first-deployment.md | | task | RCP-WP-0002-T02 | done | — | workplans/RCP-WP-0002-first-deployment.md | | task | RCP-WP-0002-T03 | done | — | workplans/RCP-WP-0002-first-deployment.md | -| task | RCP-WP-0002-T04 | progress | — | workplans/RCP-WP-0002-first-deployment.md | +| task | RCP-WP-0002-T04 | done | — | workplans/RCP-WP-0002-first-deployment.md | | task | RCP-WP-0002-T05 | wait | — | workplans/RCP-WP-0002-first-deployment.md | diff --git a/declarations/rapp.yaml b/declarations/rapp.yaml index 7856e62..d8453a2 100644 --- a/declarations/rapp.yaml +++ b/declarations/rapp.yaml @@ -4,7 +4,7 @@ rapp_id: rapp-canned-prompts repo: rapp-canned-prompts ownership_repo: canned-prompts contract_version: 1.0.0 -readiness_state: verified +readiness_state: deployed workload_identity: name: canned-prompts package_type: manifest-managed-platform-service @@ -34,11 +34,11 @@ composition: upstream_components: - name: canned-prompts source: forgejo.coulomb.social/coulomb/canned-prompts - # Published 2026-09-07 from canned-prompts service/Dockerfile, tag 0.1.5. + # Published 2026-09-07 from canned-prompts service/Dockerfile, tag 0.2.0. # Pinned by digest rather than tag: a tag can be moved, and # live-image-digest-match would then pass against something that is no # longer what this repo reviewed. - version: sha256:14c7b92f20d63f2e70ea17b0b45d3c483521fbbaf14ee5bc277fa541e10452e2 + version: sha256:d5de508ea66d3ad9e354f05ccb6f764bd3b01d290631558f0e866059e8a705a6 rollout_contract: default_mode: kubectl-server-side-apply smoke_contract: diff --git a/docs/evidence/RCP-WP-0002-T04-first-deployment-2026-09-08.md b/docs/evidence/RCP-WP-0002-T04-first-deployment-2026-09-08.md index 62439ea..a46da12 100644 --- a/docs/evidence/RCP-WP-0002-T04-first-deployment-2026-09-08.md +++ b/docs/evidence/RCP-WP-0002-T04-first-deployment-2026-09-08.md @@ -102,6 +102,7 @@ unable to distinguish their own failure from the failure they were watching for: | `check_readiness` (earlier) | one `except` reported "database unreachable" for a reachable but unmigrated database | | lease watcher v1 | empty `kubectl` output counted as an outage | | lease watcher v2 | `grep -c` exits 1 on zero matches, so `\|\| echo "?"` appended a marker to a legitimate `0` and the verdict could never read clean | +| lease watcher v3 | printed "survived past the lease TTL" after a 12-minute window on a 13-minute-old pod — it never checked the one thing it claimed | A verification step that cannot fail correctly is worse than none, because it is trusted. That is the durable lesson from this deployment, more than any single @@ -118,3 +119,19 @@ direction; the same class of bug pointing the other way is what let So the verdict logic is now itself tested — the fix was accompanied by feeding the counter a matching line and a non-matching one and confirming it returns 1 and 0 — rather than assumed correct because it looked right. + +**Version 3 then failed in the dangerous direction.** It reported +`RESULT: survived — continuously ready past the lease TTL` after a clean +12-minute window on a pod that had been up for 13 minutes. Every measurement in +it was accurate; the *conclusion* was not, because nothing checked that the +observation window exceeded the 30-minute lease it claimed to have outlasted. + +That is the failure mode worth fearing. Versions 1 and 2 cried wolf, which +provokes investigation. Version 3 would have been believed — and `verified` +recorded on it — for the same reason `live-image-digest-match` would have been +believed: a check reporting success it has not established is indistinguishable +from a check that works, right up until it matters. + +The verdict now requires `uptime > lease TTL` and reports `INCONCLUSIVE` when a +window is clean but too short, because "clean" and "proven" are different +claims and only one of them was ever being measured. diff --git a/docs/evidence/RCP-WP-0002-T05-lease-rotation-2026-09-08.log b/docs/evidence/RCP-WP-0002-T05-lease-rotation-2026-09-08.log new file mode 100644 index 0000000..3e3575b --- /dev/null +++ b/docs/evidence/RCP-WP-0002-T05-lease-rotation-2026-09-08.log @@ -0,0 +1,2 @@ +started 2026-09-08T10:48:16+02:00 — 22m watch (runtime lease TTL 30m) +2026-09-08T10:48:17+02:00 ready=1/1 diff --git a/manifests/migration.yaml b/manifests/migration.yaml index 5c78fcf..5ce49fd 100644 --- a/manifests/migration.yaml +++ b/manifests/migration.yaml @@ -5,7 +5,7 @@ apiVersion: batch/v1 kind: Job metadata: - name: canned-prompts-schema-migration-0002 + name: canned-prompts-schema-migration-0003 namespace: canned-prompts labels: app.kubernetes.io/name: canned-prompts-migration @@ -33,7 +33,7 @@ spec: type: RuntimeDefault containers: - name: migrate - image: forgejo.coulomb.social/coulomb/canned-prompts@sha256:14c7b92f20d63f2e70ea17b0b45d3c483521fbbaf14ee5bc277fa541e10452e2 + image: forgejo.coulomb.social/coulomb/canned-prompts@sha256:d5de508ea66d3ad9e354f05ccb6f764bd3b01d290631558f0e866059e8a705a6 command: ["alembic"] args: ["upgrade", "head"] workingDir: /app diff --git a/manifests/runtime.yaml b/manifests/runtime.yaml index 2bababb..1204c74 100644 --- a/manifests/runtime.yaml +++ b/manifests/runtime.yaml @@ -27,7 +27,7 @@ spec: type: RuntimeDefault containers: - name: canned-prompts - image: forgejo.coulomb.social/coulomb/canned-prompts@sha256:14c7b92f20d63f2e70ea17b0b45d3c483521fbbaf14ee5bc277fa541e10452e2 + image: forgejo.coulomb.social/coulomb/canned-prompts@sha256:d5de508ea66d3ad9e354f05ccb6f764bd3b01d290631558f0e866059e8a705a6 imagePullPolicy: IfNotPresent ports: - name: http diff --git a/tools/lease-watch.sh b/tools/lease-watch.sh index 17e3959..2be1626 100755 --- a/tools/lease-watch.sh +++ b/tools/lease-watch.sh @@ -6,7 +6,7 @@ # transient API hiccup was recorded as a service failure and produced a FAILED # verdict for a service that never returned a single 503. A check that cannot # tell its own failure from the failure it watches for is worse than no check. -OUT="$1"; MINUTES="${2:-20}" +OUT="$1"; MINUTES="${2:-20}"; LEASE_TTL_MIN="${3:-30}" DEADLINE=$(( $(date +%s) + MINUTES * 60 )) : > "$OUT" echo "started $(date -Is) — ${MINUTES}m watch (runtime lease TTL 30m)" >> "$OUT" @@ -41,14 +41,27 @@ else P503="unavailable" fi UP=$(kubectl -n canned-prompts get pod -l app.kubernetes.io/name=canned-prompts -o jsonpath='{.items[0].status.startTime}' 2>/dev/null) +UPTIME_MIN=$(python3 -c " +import datetime as dt, sys +try: + s = dt.datetime.fromisoformat('$UP'.replace('Z','+00:00')) + print(int((dt.datetime.now(dt.timezone.utc)-s).total_seconds()//60)) +except Exception: + print(-1) +" 2>/dev/null || echo -1) R=$(kubectl -n canned-prompts get pod -l app.kubernetes.io/name=canned-prompts -o jsonpath='{.items[0].status.containerStatuses[0].restartCount}' 2>/dev/null) { echo "finished $(date -Is)" echo "samples=$SAMPLES not_ready=$NOTREADY query_failed=$QUERYFAIL" echo "kubelet readiness 503s in window: $P503 (probe every 5s)" - echo "pod started $UP, restarts=$R" - if [ "$NOTREADY" -eq 0 ] && [ "$P503" = "0" ]; then - echo "RESULT: survived — continuously ready past the lease TTL" + echo "pod started $UP, uptime=${UPTIME_MIN}m, restarts=$R" + if [ "$NOTREADY" -eq 0 ] && [ "$P503" = "0" ] && [ "$UPTIME_MIN" -gt "$LEASE_TTL_MIN" ]; then + echo "RESULT: survived — continuously ready, uptime ${UPTIME_MIN}m > lease TTL ${LEASE_TTL_MIN}m" + elif [ "$NOTREADY" -eq 0 ] && [ "$P503" = "0" ]; then + # The claim is "survived a lease rotation". A clean window shorter than the + # lease does not establish it, and saying so anyway is the failure mode that + # errs toward reassurance — the one that ships. + echo "RESULT: INCONCLUSIVE — clean, but uptime ${UPTIME_MIN}m has not yet passed the ${LEASE_TTL_MIN}m lease" else echo "RESULT: FAILED" fi diff --git a/tools/smoke.sh b/tools/smoke.sh index 9adcea2..1ea7d5f 100755 --- a/tools/smoke.sh +++ b/tools/smoke.sh @@ -55,8 +55,19 @@ esac SERVICE_SMOKE=${SERVICE_SMOKE:-$HOME/canned-prompts/service/tools/smoke.py} if [ -f "$SERVICE_SMOKE" ]; then echo "--- service-level (${SERVICE_SMOKE}) ---" + # Wait for a ready endpoint before forwarding. Port-forwarding into a pod + # that is still rolling reports "connection closed" for every service-level + # check — a false failure that looks exactly like a broken service. + for _ in $(seq 1 30); do + [ "$(kubectl -n "$NS" get deploy canned-prompts -o jsonpath='{.status.readyReplicas}' 2>/dev/null)" = "1" ] && break + sleep 2 + done kubectl -n "$NS" port-forward svc/canned-prompts 18000:8000 >/dev/null 2>&1 & - PF=$!; trap 'kill $PF 2>/dev/null || true' EXIT; sleep 3 + PF=$!; trap 'kill $PF 2>/dev/null || true' EXIT + for _ in $(seq 1 15); do + curl -s -m 2 http://127.0.0.1:18000/healthz >/dev/null 2>&1 && break + sleep 1 + done python3 "$SERVICE_SMOKE" --base http://127.0.0.1:18000 --expect-migration "${EXPECT_MIGRATION:-0002}" || FAILED=$((FAILED+1)) else check "service-level-checks" "canned-prompts checkout not found at $SERVICE_SMOKE"