diff --git a/WORK-RECORDS.md b/WORK-RECORDS.md index 86302df..b78789d 100644 --- a/WORK-RECORDS.md +++ b/WORK-RECORDS.md @@ -9,12 +9,12 @@ | Kind | ID | Status | Lane | Source | | --- | --- | --- | --- | --- | | workplan | RCP-WP-0001 | finished | — | workplans/RCP-WP-0001-statehub-bootstrap.md | -| workplan | RCP-WP-0002 | finished | — | workplans/RCP-WP-0002-first-deployment.md | +| workplan | RCP-WP-0002 | active | — | workplans/RCP-WP-0002-first-deployment.md | | task | RCP-WP-0001-T01 | done | — | workplans/RCP-WP-0001-statehub-bootstrap.md | | task | RCP-WP-0001-T02 | done | — | workplans/RCP-WP-0001-statehub-bootstrap.md | | task | RCP-WP-0001-T03 | done | — | workplans/RCP-WP-0001-statehub-bootstrap.md | | task | RCP-WP-0002-T01 | done | — | workplans/RCP-WP-0002-first-deployment.md | | task | RCP-WP-0002-T02 | done | — | workplans/RCP-WP-0002-first-deployment.md | | task | RCP-WP-0002-T03 | done | — | workplans/RCP-WP-0002-first-deployment.md | -| task | RCP-WP-0002-T04 | done | — | workplans/RCP-WP-0002-first-deployment.md | +| task | RCP-WP-0002-T04 | progress | — | workplans/RCP-WP-0002-first-deployment.md | | task | RCP-WP-0002-T05 | wait | — | workplans/RCP-WP-0002-first-deployment.md | diff --git a/docs/evidence/RCP-WP-0002-T04-first-deployment-2026-09-08.md b/docs/evidence/RCP-WP-0002-T04-first-deployment-2026-09-08.md index fa0d0df..a9a0f7b 100644 --- a/docs/evidence/RCP-WP-0002-T04-first-deployment-2026-09-08.md +++ b/docs/evidence/RCP-WP-0002-T04-first-deployment-2026-09-08.md @@ -74,3 +74,34 @@ lease window cannot distinguish a service that works from one that works *once*. The criterion for `verified` is therefore observation across a full rotation, not a passing check at a single instant — an availability property is not provable by one sample. + + +## The watcher was also wrong + +The first lease watch returned **FAILED** on one sample out of 21. That sample +was the instrument, not the service: it treated an empty `kubectl` result as +"not ready", so a transient API hiccup was recorded as an outage. + +The service had returned **zero** 503s across the whole window. The readiness +probe fires every 5s, so a genuine two-minute outage would have left roughly +twenty-four of them, and the pod ran 46 minutes past a 30-minute lease with no +restarts. + +I did not override the verdict by argument — a verdict that can be talked around +is worth nothing. The watcher is fixed (`tools/lease-watch.sh`) to distinguish +a failed *query* from a failed *service*, and to corroborate its own sampling +against the kubelet's probe history, which is a far better instrument than one +sample a minute. Then re-measured. + +**Three checks in this rollout were themselves defective**, all the same shape — +unable to distinguish their own failure from the failure they were watching for: + +| Check | How it lied | +|---|---| +| `live-image-digest-match` | line-offset `grep` returned empty; reported "not pinned yet" while a digest was pinned | +| `check_readiness` (earlier) | one `except` reported "database unreachable" for a reachable but unmigrated database | +| lease watcher v1 | empty `kubectl` output counted as an outage | + +A verification step that cannot fail correctly is worse than none, because it is +trusted. That is the durable lesson from this deployment, more than any single +defect it found. diff --git a/tools/lease-watch.sh b/tools/lease-watch.sh new file mode 100755 index 0000000..7f8f5b1 --- /dev/null +++ b/tools/lease-watch.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# Verification for RCP-WP-0002-T04: stay ready across a credential lease +# rotation (runtime lease TTL 30m). +# +# v2. The first version treated an empty kubectl result as "not ready", so a +# transient API hiccup was recorded as a service failure and produced a FAILED +# verdict for a service that never returned a single 503. A check that cannot +# tell its own failure from the failure it watches for is worse than no check. +OUT="$1"; MINUTES="${2:-20}" +DEADLINE=$(( $(date +%s) + MINUTES * 60 )) +: > "$OUT" +echo "started $(date -Is) — ${MINUTES}m watch (runtime lease TTL 30m)" >> "$OUT" +NOTREADY=0; QUERYFAIL=0; SAMPLES=0 +while [ "$(date +%s)" -lt "$DEADLINE" ]; do + SAMPLES=$((SAMPLES+1)) + if ! OUTJSON=$(kubectl -n canned-prompts get deploy canned-prompts \ + -o jsonpath='{.status.readyReplicas}|{.status.replicas}' 2>/dev/null) || [ -z "${OUTJSON%%|*}${OUTJSON##*|}" ]; then + QUERYFAIL=$((QUERYFAIL+1)) + echo "$(date -Is) QUERY-FAILED (instrument, not service)" >> "$OUT" + else + READY="${OUTJSON%%|*}"; WANT="${OUTJSON##*|}" + if [ "$READY" = "$WANT" ] && [ -n "$READY" ]; then + echo "$(date -Is) ready=$READY/$WANT" >> "$OUT" + else + NOTREADY=$((NOTREADY+1)) + echo "$(date -Is) NOT READY ready=${READY:-0}/${WANT:-?}" >> "$OUT" + fi + fi + sleep 60 +done +# Corroborate with the kubelet, which probes every 5s and is a far better +# instrument than this loop's 1/min sampling. +P503=$(kubectl -n canned-prompts logs deploy/canned-prompts --since="${MINUTES}m" 2>/dev/null | grep -c '" 503' || echo "?") +UP=$(kubectl -n canned-prompts get pod -l app.kubernetes.io/name=canned-prompts -o jsonpath='{.items[0].status.startTime}' 2>/dev/null) +R=$(kubectl -n canned-prompts get pod -l app.kubernetes.io/name=canned-prompts -o jsonpath='{.items[0].status.containerStatuses[0].restartCount}' 2>/dev/null) +{ + echo "finished $(date -Is)" + echo "samples=$SAMPLES not_ready=$NOTREADY query_failed=$QUERYFAIL" + echo "kubelet readiness 503s in window: $P503 (probe every 5s)" + echo "pod started $UP, restarts=$R" + if [ "$NOTREADY" -eq 0 ] && [ "$P503" = "0" ]; then + echo "RESULT: survived — continuously ready past the lease TTL" + else + echo "RESULT: FAILED" + fi +} >> "$OUT"