2026-09-08 09:48:06 +02:00
#!/usr/bin/env bash
# Verification for RCP-WP-0002-T04: stay ready across a credential lease
# rotation (runtime lease TTL 30m).
#
# v2. The first version treated an empty kubectl result as "not ready", so a
# transient API hiccup was recorded as a service failure and produced a FAILED
# verdict for a service that never returned a single 503. A check that cannot
# tell its own failure from the failure it watches for is worse than no check.
2026-09-08 10:48:35 +02:00
OUT = " $1 " ; MINUTES = " ${ 2 :- 20 } " ; LEASE_TTL_MIN = " ${ 3 :- 30 } "
2026-09-08 09:48:06 +02:00
DEADLINE = $(( $( date +%s) + MINUTES * 60 ))
: > " $OUT "
echo " started $( date -Is) — ${ MINUTES } m watch (runtime lease TTL 30m) " >> " $OUT "
NOTREADY = 0; QUERYFAIL = 0; SAMPLES = 0
while [ " $( date +%s) " -lt " $DEADLINE " ] ; do
SAMPLES = $(( SAMPLES+1))
if ! OUTJSON = $( kubectl -n canned-prompts get deploy canned-prompts \
-o jsonpath = '{.status.readyReplicas}|{.status.replicas}' 2>/dev/null) || [ -z " ${ OUTJSON %%|* } ${ OUTJSON ##*| } " ] ; then
QUERYFAIL = $(( QUERYFAIL+1))
echo " $( date -Is) QUERY-FAILED (instrument, not service) " >> " $OUT "
else
READY = " ${ OUTJSON %%|* } " ; WANT = " ${ OUTJSON ##*| } "
if [ " $READY " = " $WANT " ] && [ -n " $READY " ] ; then
echo " $( date -Is) ready= $READY / $WANT " >> " $OUT "
else
NOTREADY = $(( NOTREADY+1))
echo " $( date -Is) NOT READY ready= ${ READY :- 0 } / ${ WANT :- ? } " >> " $OUT "
fi
fi
sleep 60
done
# Corroborate with the kubelet, which probes every 5s and is a far better
# instrument than this loop's 1/min sampling.
2026-09-08 10:09:35 +02:00
# grep -c exits 1 when the count is zero, so `|| echo "?"` appended a marker on
# top of a legitimate "0" and the verdict test never matched. Capture the logs
# first: a kubectl failure is then distinguishable from an absence of 503s,
# which is the whole point.
if LOGS = $( kubectl -n canned-prompts logs deploy/canned-prompts --since= " ${ MINUTES } m " 2>/dev/null) ; then
P503 = $( printf '%s' " $LOGS " | grep -c '" 503' || true )
P503 = ${ P503 :- 0 }
else
P503 = "unavailable"
fi
2026-09-08 09:48:06 +02:00
UP = $( kubectl -n canned-prompts get pod -l app.kubernetes.io/name= canned-prompts -o jsonpath = '{.items[0].status.startTime}' 2>/dev/null)
2026-09-08 10:48:35 +02:00
UPTIME_MIN = $( python3 -c "
import datetime as dt, sys
try:
s = dt.datetime.fromisoformat( '$UP' .replace( 'Z' ,'+00:00' ) )
print( int( ( dt.datetime.now( dt.timezone.utc) -s) .total_seconds( ) //60) )
except Exception:
print( -1)
" 2>/dev/null || echo -1)
2026-09-08 09:48:06 +02:00
R = $( kubectl -n canned-prompts get pod -l app.kubernetes.io/name= canned-prompts -o jsonpath = '{.items[0].status.containerStatuses[0].restartCount}' 2>/dev/null)
{
echo " finished $( date -Is) "
echo " samples= $SAMPLES not_ready= $NOTREADY query_failed= $QUERYFAIL "
echo " kubelet readiness 503s in window: $P503 (probe every 5s) "
2026-09-08 10:48:35 +02:00
echo " pod started $UP , uptime= ${ UPTIME_MIN } m, restarts= $R "
if [ " $NOTREADY " -eq 0 ] && [ " $P503 " = "0" ] && [ " $UPTIME_MIN " -gt " $LEASE_TTL_MIN " ] ; then
echo " RESULT: survived — continuously ready, uptime ${ UPTIME_MIN } m > lease TTL ${ LEASE_TTL_MIN } m "
elif [ " $NOTREADY " -eq 0 ] && [ " $P503 " = "0" ] ; then
# The claim is "survived a lease rotation". A clean window shorter than the
# lease does not establish it, and saying so anyway is the failure mode that
# errs toward reassurance — the one that ships.
echo " RESULT: INCONCLUSIVE — clean, but uptime ${ UPTIME_MIN } m has not yet passed the ${ LEASE_TTL_MIN } m lease "
2026-09-08 09:48:06 +02:00
else
echo "RESULT: FAILED"
fi
} >> " $OUT "