Version 3 reported 'survived past the lease TTL' after a clean 12-minute window on a 13-minute-old pod. Every measurement in it was accurate; the conclusion was not, because nothing checked that the observation window had actually exceeded the 30-minute lease it claimed to have outlasted. Fifth defect in this instrument, and the first to err toward reassurance. Versions 1 and 2 cried wolf, which provokes investigation. This one would have been believed, and readiness_state: verified recorded on it — the same way live-image-digest-match would have been believed. A check reporting success it has not established is indistinguishable from one that works, until it matters. The verdict now requires uptime > lease TTL and reports INCONCLUSIVE when a window is clean but too short. 'Clean' and 'proven' are different claims and only one of them was being measured. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Bjefh8NUiEiahN4JLwoSKM Assistant: claude-code Assistant-Model: opus Assistant-Process: 388925@bnt-lap001 Assistant-Session: 3507023f-e0fd-4a1e-9d90-a0d4217d1502
77 lines
3.8 KiB
Bash
Executable file
77 lines
3.8 KiB
Bash
Executable file
#!/usr/bin/env bash
|
|
# Deployment-level smoke checks for rapp-canned-prompts.
|
|
#
|
|
# The service-level half lives in canned-prompts (`service/tools/smoke.py`) and
|
|
# is called from here rather than reimplemented — the deployment should not hold
|
|
# a second opinion about whether the service is healthy.
|
|
#
|
|
# What is checked here is what only the cluster can answer: that secrets
|
|
# materialized, that the Service is private, that NetworkPolicies exist, and
|
|
# that the running image is the digest this repo pins.
|
|
set -euo pipefail
|
|
|
|
NS=${NS:-canned-prompts}
|
|
# Parsed, not grepped. A line-offset grep silently returned empty once comments
|
|
# were added above `version:`, and the check then degraded to "not pinned yet"
|
|
# instead of failing — it could not have passed for any pin.
|
|
EXPECT_DIGEST=${EXPECT_DIGEST:-$(python3 -c "
|
|
import yaml, sys
|
|
d = yaml.safe_load(open('declarations/rapp.yaml'))
|
|
c = [u for u in d['composition']['upstream_components'] if u['name'] == 'canned-prompts']
|
|
print(c[0]['version'] if c else '')
|
|
" 2>/dev/null || true)}
|
|
FAILED=0
|
|
|
|
check() { # name, condition-output
|
|
if [ "$2" = "ok" ]; then printf 'PASS %s\n' "$1"; else printf 'FAIL %s %s\n' "$1" "$2"; FAILED=$((FAILED+1)); fi
|
|
}
|
|
|
|
# external-secrets-ready
|
|
for s in canned-prompts-postgres-runtime canned-prompts-postgres-migration; do
|
|
if kubectl -n "$NS" get secret "$s" >/dev/null 2>&1; then check "external-secrets-ready:$s" ok
|
|
else check "external-secrets-ready:$s" "secret absent"; fi
|
|
done
|
|
|
|
# private-service-only — a ClusterIP and no Ingress pointing at it
|
|
TYPE=$(kubectl -n "$NS" get svc canned-prompts -o jsonpath='{.spec.type}' 2>/dev/null || echo missing)
|
|
[ "$TYPE" = "ClusterIP" ] && check "private-service-only:type" ok || check "private-service-only:type" "type=$TYPE"
|
|
INGRESS=$(kubectl -n "$NS" get ingress -o name 2>/dev/null | wc -l)
|
|
[ "$INGRESS" = "0" ] && check "private-service-only:no-ingress" ok || check "private-service-only:no-ingress" "$INGRESS ingress objects"
|
|
|
|
# networkpolicies-present — a default-deny plus the runtime policy
|
|
NP=$(kubectl -n "$NS" get networkpolicy -o name 2>/dev/null | wc -l)
|
|
[ "$NP" -ge 2 ] && check "networkpolicies-present" ok || check "networkpolicies-present" "$NP policies"
|
|
|
|
# live-image-digest-match
|
|
LIVE=$(kubectl -n "$NS" get deploy canned-prompts -o jsonpath='{.spec.template.spec.containers[0].image}' 2>/dev/null || echo missing)
|
|
case "$EXPECT_DIGEST" in
|
|
""|pending-publication)
|
|
check "live-image-digest-match" "no digest pinned yet (declarations/rapp.yaml says pending-publication)" ;;
|
|
*) [ "${LIVE##*@}" = "$EXPECT_DIGEST" ] && check "live-image-digest-match" ok \
|
|
|| check "live-image-digest-match" "live=$LIVE expected=$EXPECT_DIGEST" ;;
|
|
esac
|
|
|
|
# state-health-ok and migration-at-head, from the owning repo's checker.
|
|
SERVICE_SMOKE=${SERVICE_SMOKE:-$HOME/canned-prompts/service/tools/smoke.py}
|
|
if [ -f "$SERVICE_SMOKE" ]; then
|
|
echo "--- service-level (${SERVICE_SMOKE}) ---"
|
|
# Wait for a ready endpoint before forwarding. Port-forwarding into a pod
|
|
# that is still rolling reports "connection closed" for every service-level
|
|
# check — a false failure that looks exactly like a broken service.
|
|
for _ in $(seq 1 30); do
|
|
[ "$(kubectl -n "$NS" get deploy canned-prompts -o jsonpath='{.status.readyReplicas}' 2>/dev/null)" = "1" ] && break
|
|
sleep 2
|
|
done
|
|
kubectl -n "$NS" port-forward svc/canned-prompts 18000:8000 >/dev/null 2>&1 &
|
|
PF=$!; trap 'kill $PF 2>/dev/null || true' EXIT
|
|
for _ in $(seq 1 15); do
|
|
curl -s -m 2 http://127.0.0.1:18000/healthz >/dev/null 2>&1 && break
|
|
sleep 1
|
|
done
|
|
python3 "$SERVICE_SMOKE" --base http://127.0.0.1:18000 --expect-migration "${EXPECT_MIGRATION:-0002}" || FAILED=$((FAILED+1))
|
|
else
|
|
check "service-level-checks" "canned-prompts checkout not found at $SERVICE_SMOKE"
|
|
fi
|
|
|
|
[ "$FAILED" -eq 0 ] || { echo; echo "$FAILED check(s) failed" >&2; exit 1; }
|
|
echo; echo "all deployment checks passed"
|