#!/usr/bin/env bash # Verification for RCP-WP-0002-T04: stay ready across a credential lease # rotation (runtime lease TTL 30m). # # v2. The first version treated an empty kubectl result as "not ready", so a # transient API hiccup was recorded as a service failure and produced a FAILED # verdict for a service that never returned a single 503. A check that cannot # tell its own failure from the failure it watches for is worse than no check. OUT="$1"; MINUTES="${2:-20}"; LEASE_TTL_MIN="${3:-30}" DEADLINE=$(( $(date +%s) + MINUTES * 60 )) : > "$OUT" echo "started $(date -Is) — ${MINUTES}m watch (runtime lease TTL 30m)" >> "$OUT" NOTREADY=0; QUERYFAIL=0; SAMPLES=0 while [ "$(date +%s)" -lt "$DEADLINE" ]; do SAMPLES=$((SAMPLES+1)) if ! OUTJSON=$(kubectl -n canned-prompts get deploy canned-prompts \ -o jsonpath='{.status.readyReplicas}|{.status.replicas}' 2>/dev/null) || [ -z "${OUTJSON%%|*}${OUTJSON##*|}" ]; then QUERYFAIL=$((QUERYFAIL+1)) echo "$(date -Is) QUERY-FAILED (instrument, not service)" >> "$OUT" else READY="${OUTJSON%%|*}"; WANT="${OUTJSON##*|}" if [ "$READY" = "$WANT" ] && [ -n "$READY" ]; then echo "$(date -Is) ready=$READY/$WANT" >> "$OUT" else NOTREADY=$((NOTREADY+1)) echo "$(date -Is) NOT READY ready=${READY:-0}/${WANT:-?}" >> "$OUT" fi fi sleep 60 done # Corroborate with the kubelet, which probes every 5s and is a far better # instrument than this loop's 1/min sampling. # grep -c exits 1 when the count is zero, so `|| echo "?"` appended a marker on # top of a legitimate "0" and the verdict test never matched. Capture the logs # first: a kubectl failure is then distinguishable from an absence of 503s, # which is the whole point. if LOGS=$(kubectl -n canned-prompts logs deploy/canned-prompts --since="${MINUTES}m" 2>/dev/null); then P503=$(printf '%s' "$LOGS" | grep -c '" 503' || true) P503=${P503:-0} else P503="unavailable" fi UP=$(kubectl -n canned-prompts get pod -l app.kubernetes.io/name=canned-prompts -o jsonpath='{.items[0].status.startTime}' 2>/dev/null) UPTIME_MIN=$(python3 -c " import datetime as dt, sys try: s = dt.datetime.fromisoformat('$UP'.replace('Z','+00:00')) print(int((dt.datetime.now(dt.timezone.utc)-s).total_seconds()//60)) except Exception: print(-1) " 2>/dev/null || echo -1) R=$(kubectl -n canned-prompts get pod -l app.kubernetes.io/name=canned-prompts -o jsonpath='{.items[0].status.containerStatuses[0].restartCount}' 2>/dev/null) { echo "finished $(date -Is)" echo "samples=$SAMPLES not_ready=$NOTREADY query_failed=$QUERYFAIL" echo "kubelet readiness 503s in window: $P503 (probe every 5s)" echo "pod started $UP, uptime=${UPTIME_MIN}m, restarts=$R" if [ "$NOTREADY" -eq 0 ] && [ "$P503" = "0" ] && [ "$UPTIME_MIN" -gt "$LEASE_TTL_MIN" ]; then echo "RESULT: survived — continuously ready, uptime ${UPTIME_MIN}m > lease TTL ${LEASE_TTL_MIN}m" elif [ "$NOTREADY" -eq 0 ] && [ "$P503" = "0" ]; then # The claim is "survived a lease rotation". A clean window shorter than the # lease does not establish it, and saying so anyway is the failure mode that # errs toward reassurance — the one that ships. echo "RESULT: INCONCLUSIVE — clean, but uptime ${UPTIME_MIN}m has not yet passed the ${LEASE_TTL_MIN}m lease" else echo "RESULT: FAILED" fi } >> "$OUT"