who_need_help/scripts/kind-rolling-verify.sh

418 lines
12 KiB
Bash
Executable File

#!/usr/bin/env bash
set -euo pipefail
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)
KUBECTL="$ROOT/.tools/bin/kubectl"
CONTEXT=kind-who-need-help
CLUSTER_CONTAINER=who-need-help-control-plane
NAMESPACE=who-need-help
WEB_DEPLOYMENT=who-need-help-who-need-help-web
WORKER_DEPLOYMENT=who-need-help-who-need-help-worker
SERVICE=who-need-help-who-need-help
OWNERSHIP_MARKER="$ROOT/.tools/who-need-help.owned"
ROLLOUT_TIMEOUT=${KIND_ROLLOUT_TIMEOUT:-180s}
PROBE_INTERVAL=${KIND_ROLLOUT_PROBE_INTERVAL_SECONDS:-0.05}
PROBE_TIMEOUT=${KIND_ROLLOUT_PROBE_TIMEOUT_SECONDS:-2}
PROBE_RETRIES=${KIND_ROLLOUT_PROBE_RETRIES:-2}
CLUSTER_JOIN_TIMEOUT=${KIND_CLUSTER_JOIN_TIMEOUT_SECONDS:-120}
LABEL=${1:-"kind-rollout-$(date -u +%Y%m%dT%H%M%SZ)"}
if [[ ! -x "$KUBECTL" ]]; then
echo "The project-owned kubectl is missing. Run scripts/bootstrap-kubernetes-tools.sh." >&2
exit 1
fi
if [[ ! -f "$OWNERSHIP_MARKER" ]]; then
echo "The kind ownership marker is missing; refusing to mutate the cluster." >&2
exit 1
fi
if [[ ! "$LABEL" =~ ^[A-Za-z0-9._-]+$ ]]; then
echo "Run label may contain only letters, numbers, dot, underscore, and dash." >&2
exit 1
fi
if [[ ! "$ROLLOUT_TIMEOUT" =~ ^[1-9][0-9]*[smh]$ ]]; then
echo "KIND_ROLLOUT_TIMEOUT must be a positive Kubernetes duration in s, m, or h." >&2
exit 1
fi
if ! awk -v value="$PROBE_INTERVAL" \
'BEGIN {exit !(value ~ /^[0-9]+([.][0-9]+)?$/ && value > 0)}'; then
echo "KIND_ROLLOUT_PROBE_INTERVAL_SECONDS must be greater than zero." >&2
exit 1
fi
if [[ ! "$PROBE_TIMEOUT" =~ ^[1-9][0-9]*$ ]]; then
echo "KIND_ROLLOUT_PROBE_TIMEOUT_SECONDS must be a positive integer." >&2
exit 1
fi
if [[ ! "$PROBE_RETRIES" =~ ^[0-9]+$ ]]; then
echo "KIND_ROLLOUT_PROBE_RETRIES must be a non-negative integer." >&2
exit 1
fi
if [[ ! "$CLUSTER_JOIN_TIMEOUT" =~ ^[1-9][0-9]*$ ]]; then
echo "KIND_CLUSTER_JOIN_TIMEOUT_SECONDS must be a positive integer." >&2
exit 1
fi
kube=("$KUBECTL" --context "$CONTEXT" --namespace "$NAMESPACE")
if ! "$KUBECTL" config get-contexts -o name | grep -Fxq "$CONTEXT"; then
echo "The expected local kind context does not exist." >&2
exit 1
fi
if [[ "$(docker inspect --format '{{index .Config.Labels "io.x-k8s.kind.cluster"}}' \
"$CLUSTER_CONTAINER")" != "who-need-help" ]]; then
echo "The kind control-plane container does not belong to this project cluster." >&2
exit 1
fi
output_dir="$ROOT/output/resilience/$LABEL"
mkdir -p "$output_dir"
chmod 700 "$ROOT/output" "$ROOT/output/resilience" "$output_dir"
probe_marker="$output_dir/.probe-running"
probe_log="$output_dir/readiness.jsonl"
run_started_at=$(date -u +%Y-%m-%dT%H:%M:%SZ)
touch "$probe_marker"
deployment_snapshot() {
"${kube[@]}" get deployment "$WEB_DEPLOYMENT" "$WORKER_DEPLOYMENT" -o json |
jq '[
.items[] | {
name: .metadata.name,
generation: .metadata.generation,
revision: .metadata.annotations["deployment.kubernetes.io/revision"],
replicas: .spec.replicas,
ready: .status.readyReplicas,
available: .status.availableReplicas,
updated: .status.updatedReplicas,
strategy: .spec.strategy,
image: .spec.template.spec.containers[0].image
}
]' >"$1"
}
database_snapshot() {
# Variables are intentionally expanded inside the PostGIS container.
# shellcheck disable=SC2016
"${kube[@]}" exec -i postgis-0 -- sh -c \
'psql --no-psqlrc --tuples-only --no-align --set ON_ERROR_STOP=1 \
--username "$POSTGRES_USER" --dbname "$POSTGRES_DB"' >"$1" <<'SQL'
BEGIN READ ONLY;
SELECT 'users' AS table_name, count(*) AS row_count FROM users
UNION ALL SELECT 'users_tokens', count(*) FROM users_tokens
UNION ALL SELECT 'help_requests', count(*) FROM help_requests
UNION ALL SELECT 'messages', count(*) FROM messages
UNION ALL SELECT 'categories', count(*) FROM categories
UNION ALL SELECT 'help_assignments', count(*) FROM help_assignments
UNION ALL SELECT 'activities', count(*) FROM activities
UNION ALL SELECT 'reports', count(*) FROM reports
UNION ALL SELECT 'social_identities', count(*) FROM social_identities
UNION ALL SELECT 'tracking_sessions', count(*) FROM tracking_sessions
UNION ALL SELECT 'tracking_positions', count(*) FROM tracking_positions
UNION ALL SELECT 'schema_migrations', count(*) FROM schema_migrations
ORDER BY table_name;
COMMIT;
SQL
}
pod_snapshot() {
"${kube[@]}" get pods \
-l app.kubernetes.io/instance=who-need-help \
-o json |
jq '[
.items[]
| select(
.metadata.labels["app.kubernetes.io/component"] == "web" or
.metadata.labels["app.kubernetes.io/component"] == "worker"
)
| {
name: .metadata.name,
uid: .metadata.uid,
component: .metadata.labels["app.kubernetes.io/component"],
ready: ([.status.containerStatuses[]?.ready] | all),
restarts: ([.status.containerStatuses[]?.restartCount] | add // 0)
}
] | sort_by(.component, .name)' >"$1"
}
stop_probe() {
unlink "$probe_marker" 2>/dev/null || true
if [[ -n "${probe_pid:-}" ]]; then
wait "$probe_pid" 2>/dev/null || true
fi
}
cleanup() {
local status=$?
trap - EXIT HUP INT TERM
stop_probe
exit "$status"
}
trap cleanup EXIT HUP INT TERM
deployment_snapshot "$output_dir/deployments-before.json"
if ! jq -e '
length == 2 and
all(
.replicas >= 2 and
.ready == .replicas and
.available == .replicas and
.updated == .replicas and
.strategy.type == "RollingUpdate" and
.strategy.rollingUpdate.maxUnavailable == 0 and
.strategy.rollingUpdate.maxSurge == 1
)
' "$output_dir/deployments-before.json" >/dev/null; then
echo "The local deployments are not ready for the recorded rolling strategy." >&2
exit 1
fi
database_snapshot "$output_dir/database-before.txt"
pod_snapshot "$output_dir/pods-before.json"
node_port=$(
"${kube[@]}" get service "$SERVICE" \
-o jsonpath='{.spec.ports[?(@.name=="http")].nodePort}'
)
published=$(
docker port "$CLUSTER_CONTAINER" "${node_port}/tcp" |
head -n 1
)
host_port=${published##*:}
if [[ ! "$host_port" =~ ^[1-9][0-9]*$ ]]; then
echo "The kind HTTP NodePort has no observed Docker host mapping." >&2
exit 1
fi
base_url="http://127.0.0.1:$host_port"
sample_readiness() {
while [[ -e "$probe_marker" ]]; do
observed_at=$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ)
body_file="$output_dir/kind-readiness-body.$$"
error_file="$output_dir/kind-readiness-error.$$"
set +e
result=$(
curl --silent --show-error \
--max-time "$PROBE_TIMEOUT" \
--retry "$PROBE_RETRIES" \
--retry-all-errors \
--retry-connrefused \
--retry-delay 0 \
--output "$body_file" \
--write-out '%{http_code} %{num_retries}' \
"$base_url/healthz/ready" 2>"$error_file"
)
curl_status=$?
set -e
read -r status retries <<<"$result"
if [[ "$curl_status" -ne 0 ]]; then
status=000
fi
if [[ -f "$body_file" ]]; then
body=$(tr -d '\n' <"$body_file")
else
body=
fi
error=$(tr -d '\n' <"$error_file")
unlink "$body_file" 2>/dev/null || true
unlink "$error_file" 2>/dev/null || true
jq -cn \
--arg observed_at "$observed_at" \
--arg status "$status" \
--arg body "$body" \
--arg error "$error" \
--argjson retries "${retries:-0}" \
'{
observed_at: $observed_at,
status: $status,
retries: $retries,
body: $body,
error: $error
}' >>"$probe_log"
sleep "$PROBE_INTERVAL"
done
}
{
printf 'observed_at=%s\n' "$run_started_at"
printf 'context=%s\n' "$CONTEXT"
printf 'namespace=%s\n' "$NAMESPACE"
printf 'rollout_timeout=%s\n' "$ROLLOUT_TIMEOUT"
printf 'probe_interval_seconds=%s\n' "$PROBE_INTERVAL"
printf 'probe_timeout_seconds=%s\n' "$PROBE_TIMEOUT"
printf 'probe_retries=%s\n' "$PROBE_RETRIES"
printf 'cluster_join_timeout_seconds=%s\n' "$CLUSTER_JOIN_TIMEOUT"
printf 'node_port=%s\n' "$node_port"
printf 'observed_host_port=%s\n' "$host_port"
"$KUBECTL" version --client
} >"$output_dir/environment.txt"
sample_readiness &
probe_pid=$!
"${kube[@]}" rollout restart "deployment/$WEB_DEPLOYMENT" \
>"$output_dir/web-rollout-restart.txt"
"${kube[@]}" rollout status "deployment/$WEB_DEPLOYMENT" \
"--timeout=$ROLLOUT_TIMEOUT" >"$output_dir/web-rollout-status.txt"
"${kube[@]}" rollout restart "deployment/$WORKER_DEPLOYMENT" \
>"$output_dir/worker-rollout-restart.txt"
"${kube[@]}" rollout status "deployment/$WORKER_DEPLOYMENT" \
"--timeout=$ROLLOUT_TIMEOUT" >"$output_dir/worker-rollout-status.txt"
"${kube[@]}" wait \
--for=condition=Ready \
--timeout="$ROLLOUT_TIMEOUT" \
pod \
-l app.kubernetes.io/instance=who-need-help \
>"$output_dir/pods-ready.txt"
old_pod_deadline=$((SECONDS + CLUSTER_JOIN_TIMEOUT))
while ((SECONDS < old_pod_deadline)); do
"${kube[@]}" get pods \
-l app.kubernetes.io/instance=who-need-help \
-o json >"$output_dir/pods-current.json"
retained_count=$(
jq \
--slurpfile before "$output_dir/pods-before.json" \
'[
.items[].metadata.uid as $uid
| select($before[0] | any(.uid == $uid))
] | length' "$output_dir/pods-current.json"
)
if [[ "$retained_count" == "0" ]]; then
break
fi
sleep 1
done
unlink "$output_dir/pods-current.json" 2>/dev/null || true
if [[ "$retained_count" != "0" ]]; then
echo "Old application pods did not terminate after the rollout." >&2
exit 1
fi
deployment_snapshot "$output_dir/deployments-after.json"
pod_snapshot "$output_dir/pods-after.json"
database_snapshot "$output_dir/database-after.txt"
if ! diff -u "$output_dir/database-before.txt" "$output_dir/database-after.txt" \
>"$output_dir/database-diff.txt"; then
echo "The rolling verification changed tracked database counts." >&2
exit 1
fi
jq -n \
--slurpfile before "$output_dir/pods-before.json" \
--slurpfile after "$output_dir/pods-after.json" \
'{
before_count: ($before[0] | length),
after_count: ($after[0] | length),
retained_uids: (
[$before[0][].uid] as $old
| [$after[0][].uid | select(. as $uid | $old | index($uid))]
),
after_all_ready: ($after[0] | all(.ready and .restarts == 0))
}' >"$output_dir/pod-replacement-summary.json"
if ! jq -e '
.before_count == 4 and
.after_count == 4 and
.retained_uids == [] and
.after_all_ready
' "$output_dir/pod-replacement-summary.json" >/dev/null; then
echo "The rollout did not replace all four application pods cleanly." >&2
exit 1
fi
web_pod=$(
"${kube[@]}" get pod \
-l app.kubernetes.io/component=web \
--field-selector=status.phase=Running \
-o jsonpath='{.items[0].metadata.name}'
)
expected_peers=$(
jq '[.[].replicas] | add - 1' "$output_dir/deployments-after.json"
)
cluster_deadline=$((SECONDS + CLUSTER_JOIN_TIMEOUT))
while ((SECONDS < cluster_deadline)); do
peer_count=$(
"${kube[@]}" exec "$web_pod" -- /app/bin/who_need_help rpc \
'IO.puts(length(Node.list()))' 2>/dev/null |
tail -n 1
)
if [[ "$peer_count" == "$expected_peers" ]]; then
break
fi
sleep 1
done
if [[ "$peer_count" != "$expected_peers" ]]; then
echo "All rolled application pods did not join the BEAM cluster." >&2
exit 1
fi
"${kube[@]}" exec "$web_pod" -- /app/bin/who_need_help rpc \
'IO.inspect(%{node: node(), peers: Node.list(), peer_count: length(Node.list())})' \
>"$output_dir/cluster-after.txt"
"$ROOT/scripts/verify-realtime-cluster.sh" kind \
>"$output_dir/pubsub-after.txt"
stop_probe
probe_pid=
jq -s '{
samples: length,
failures: (map(select(.status != "200")) | length),
retried_samples: (map(select(.retries > 0)) | length),
total_retries: (map(.retries) | add),
nodes: (map(.body | fromjson? | .node) | map(select(. != null)) | unique)
}' "$probe_log" >"$output_dir/readiness-summary.json"
"${kube[@]}" get pods \
-l app.kubernetes.io/component=web \
-o json |
jq '[.items[].status.podIP | "who_need_help@" + .] | sort' \
>"$output_dir/current-web-nodes.json"
if ! jq -e \
--slurpfile current "$output_dir/current-web-nodes.json" '
.samples > 0 and
.failures == 0 and
($current[0] | length) == 2 and
(($current[0] - .nodes) | length) == 0
' "$output_dir/readiness-summary.json" >/dev/null; then
echo "Kind readiness was unavailable or did not reach both web replicas." >&2
exit 1
fi
"${kube[@]}" logs \
--selector app.kubernetes.io/instance=who-need-help \
--all-containers \
--since-time "$run_started_at" \
--prefix >"$output_dir/application.log" 2>&1
trap - EXIT HUP INT TERM
printf 'Kind rollout evidence: %s\n' "$output_dir"