who_need_help/scripts/load-resilience-run.sh

457 lines
13 KiB
Bash
Executable File

#!/usr/bin/env bash
set -euo pipefail
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)
ENV_FILE="$ROOT/.env.load"
LABEL=${1:-"resilience-$(date -u +%Y%m%dT%H%M%SZ)"}
if [[ ! -f "$ENV_FILE" ]]; then
echo "Missing $ENV_FILE. Run scripts/ensure-local-load-env.sh first." >&2
exit 1
fi
set -a
# shellcheck source=/dev/null
. "$ENV_FILE"
set +a
for name in LOAD_PROJECT LOAD_HOST LOAD_WEB_REPLICAS LOAD_WORKER_REPLICAS \
LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS \
LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS \
LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS TRAEFIK_RETRY_ATTEMPTS \
HTTP_PORT POSTGRES_DB; do
if [[ -z "${!name:-}" ]]; then
echo "$name is missing from .env.load" >&2
exit 1
fi
done
if [[ "$LOAD_PROJECT" == "who_need_help" ]]; then
echo "The resilience profile must not use the staging Compose project." >&2
exit 1
fi
if [[ ! "$LABEL" =~ ^[A-Za-z0-9._-]+$ ]]; then
echo "Run label may contain only letters, numbers, dot, underscore, and dash." >&2
exit 1
fi
for name in LOAD_WEB_REPLICAS LOAD_WORKER_REPLICAS \
LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS \
LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS TRAEFIK_RETRY_ATTEMPTS; do
if [[ ! "${!name}" =~ ^[1-9][0-9]*$ ]]; then
echo "$name must be a positive integer." >&2
exit 1
fi
done
if ! awk -v value="$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS" \
'BEGIN {exit !(value ~ /^[0-9]+([.][0-9]+)?$/ && value > 0)}'; then
echo "LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS must be greater than zero." >&2
exit 1
fi
compose=(
docker compose
--env-file "$ENV_FILE"
-p "$LOAD_PROJECT"
-f compose.yaml
-f compose.load.yaml
)
output_dir="$ROOT/output/resilience/$LABEL"
mkdir -p "$output_dir"
chmod 700 "$ROOT/output" "$ROOT/output/resilience" "$output_dir"
run_started_at=$(date -u +%Y-%m-%dT%H:%M:%SZ)
probe_marker="$output_dir/.probe-running"
probe_log="$output_dir/readiness.jsonl"
touch "$probe_marker"
assert_scope() {
local container_id=$1
local expected_service=$2
local observed_project observed_service
observed_project=$(
docker inspect --format '{{index .Config.Labels "com.docker.compose.project"}}' \
"$container_id"
)
observed_service=$(
docker inspect --format '{{index .Config.Labels "com.docker.compose.service"}}' \
"$container_id"
)
if [[ "$observed_project" != "$LOAD_PROJECT" ||
"$observed_service" != "$expected_service" ]]; then
echo "Container scope mismatch for $container_id." >&2
exit 1
fi
}
service_ids() {
"${compose[@]}" ps -q "$1"
}
wait_for_service() {
local service=$1
local expected=$2
local require_health=$3
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
while ((SECONDS < deadline)); do
mapfile -t ids < <(service_ids "$service")
if [[ "${#ids[@]}" -eq "$expected" ]]; then
local ready=true
local id state health
for id in "${ids[@]}"; do
state=$(docker inspect --format '{{.State.Status}}' "$id")
health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{end}}' "$id")
if [[ "$state" != "running" ||
("$require_health" == "true" && "$health" != "healthy") ]]; then
ready=false
break
fi
done
if [[ "$ready" == "true" ]]; then
return 0
fi
fi
sleep 1
done
echo "Timed out waiting for $expected ready $service replicas." >&2
return 1
}
wait_for_restart() {
local container_id=$1
local previous_restarts=$2
local require_health=$3
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
while ((SECONDS < deadline)); do
local state restarts health
state=$(docker inspect --format '{{.State.Status}}' "$container_id")
restarts=$(docker inspect --format '{{.RestartCount}}' "$container_id")
health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{end}}' "$container_id")
if [[ "$state" == "running" && "$restarts" -gt "$previous_restarts" &&
("$require_health" == "false" || "$health" == "healthy") ]]; then
return 0
fi
sleep 1
done
echo "Timed out waiting for container $container_id to restart." >&2
return 1
}
wait_for_cluster() {
local expected_peers=$((LOAD_WEB_REPLICAS + LOAD_WORKER_REPLICAS - 1))
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
while ((SECONDS < deadline)); do
local target peer_count
target=$(service_ids web | head -n 1)
assert_scope "$target" web
peer_count=$(
docker exec "$target" /app/bin/who_need_help rpc \
'IO.puts(length(Node.list()))' 2>/dev/null |
tail -n 1
)
if [[ "$peer_count" == "$expected_peers" ]]; then
docker exec "$target" /app/bin/who_need_help rpc \
'IO.inspect(%{node: node(), peers: Node.list(), peer_count: length(Node.list())})'
return 0
fi
sleep 1
done
echo "Timed out waiting for all local BEAM replicas to join the cluster." >&2
return 1
}
sample_readiness() {
while [[ -e "$probe_marker" ]]; do
observed_at=$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ)
body_file="$output_dir/readiness-body.$$"
error_file="$output_dir/readiness-error.$$"
set +e
status=$(
curl --silent --show-error \
--max-time "$LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS" \
--header "Host: $LOAD_HOST" \
--output "$body_file" \
--write-out '%{http_code}' \
"http://localhost:$HTTP_PORT/healthz/ready" 2>"$error_file"
)
curl_status=$?
set -e
if [[ "$curl_status" -ne 0 ]]; then
status=000
fi
if [[ -f "$body_file" ]]; then
body=$(tr -d '\n' <"$body_file")
else
body=
fi
error=$(tr -d '\n' <"$error_file")
unlink "$body_file" 2>/dev/null || true
unlink "$error_file" 2>/dev/null || true
jq -cn \
--arg observed_at "$observed_at" \
--arg status "$status" \
--arg body "$body" \
--arg error "$error" \
'{observed_at: $observed_at, status: $status, body: $body, error: $error}' \
>>"$probe_log"
sleep "$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS"
done
}
stop_probe() {
unlink "$probe_marker" 2>/dev/null || true
if [[ -n "${probe_pid:-}" ]]; then
wait "$probe_pid" 2>/dev/null || true
fi
}
delete_retry_job() {
local destination=$1
local target
target=$(service_ids worker | head -n 1)
assert_scope "$target" worker
docker exec "$target" /app/bin/who_need_help rpc \
"import Ecto.Query
run_id = \"$LABEL\"
{count, _} =
WhoNeedHelp.Repo.delete_all(
from job in Oban.Job,
where:
job.worker == \"WhoNeedHelp.Workers.LocalRetryProbe\" and
fragment(\"?->>'run_id'\", job.args) == ^run_id
)
IO.puts(count)" >"$destination"
}
restore_topology() {
"${compose[@]}" up -d --no-deps \
--scale "web=$LOAD_WEB_REPLICAS" \
--scale "worker=$LOAD_WORKER_REPLICAS" \
web worker >/dev/null 2>&1 || true
}
cleanup() {
local status=$?
local cleanup_status=0
trap - EXIT HUP INT TERM
stop_probe
set +e
restore_topology
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
if [[ "${retry_job_inserted:-false}" == "true" ]]; then
delete_retry_job "$output_dir/retry-job-cleanup-on-exit.txt"
cleanup_status=$?
fi
set -e
if [[ "$cleanup_status" -ne 0 ]]; then
echo "Resilience probe cleanup failed." >&2
status=1
fi
exit "$status"
}
trap cleanup EXIT HUP INT TERM
wait_for_service web "$LOAD_WEB_REPLICAS" true
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
{
printf 'observed_at=%s\n' "$run_started_at"
printf 'load_project=%s\n' "$LOAD_PROJECT"
printf 'web_replicas=%s\n' "$LOAD_WEB_REPLICAS"
printf 'worker_replicas=%s\n' "$LOAD_WORKER_REPLICAS"
printf 'recovery_timeout_seconds=%s\n' "$LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS"
printf 'probe_interval_seconds=%s\n' "$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS"
printf 'request_timeout_seconds=%s\n' "$LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS"
printf 'traefik_retry_attempts=%s\n' "$TRAEFIK_RETRY_ATTEMPTS"
docker compose version
docker info --format 'docker_server={{.ServerVersion}}'
} >"$output_dir/environment.txt"
sample_readiness &
probe_pid=$!
web_target=$(service_ids web | head -n 1)
assert_scope "$web_target" web
web_restarts=$(docker inspect --format '{{.RestartCount}}' "$web_target")
docker exec "$web_target" /app/bin/who_need_help rpc ':erlang.halt(1)' \
>"$output_dir/web-kill.txt" 2>&1 || true
wait_for_restart "$web_target" "$web_restarts" true
docker inspect "$web_target" |
jq '.[0] | {
id: .Id,
project: .Config.Labels["com.docker.compose.project"],
service: .Config.Labels["com.docker.compose.service"],
restart_count: .RestartCount,
state: .State.Status,
health: .State.Health.Status
}' >"$output_dir/web-restart.json"
worker_target=$(service_ids worker | head -n 1)
assert_scope "$worker_target" worker
worker_restarts=$(docker inspect --format '{{.RestartCount}}' "$worker_target")
docker exec "$worker_target" /app/bin/who_need_help rpc ':erlang.halt(1)' \
>"$output_dir/worker-kill.txt" 2>&1 || true
wait_for_restart "$worker_target" "$worker_restarts" false
docker inspect "$worker_target" |
jq '.[0] | {
id: .Id,
project: .Config.Labels["com.docker.compose.project"],
service: .Config.Labels["com.docker.compose.service"],
restart_count: .RestartCount,
state: .State.Status
}' >"$output_dir/worker-restart.json"
mapfile -t original_web_ids < <(service_ids web)
for container_id in "${original_web_ids[@]}"; do
assert_scope "$container_id" web
{
docker stop --time 30 "$container_id"
docker rm "$container_id"
"${compose[@]}" up -d --no-deps --scale "web=$LOAD_WEB_REPLICAS" web
} >>"$output_dir/web-replacements.txt"
wait_for_service web "$LOAD_WEB_REPLICAS" true
done
mapfile -t original_worker_ids < <(service_ids worker)
for container_id in "${original_worker_ids[@]}"; do
assert_scope "$container_id" worker
{
docker stop --time 30 "$container_id"
docker rm "$container_id"
"${compose[@]}" up -d --no-deps --scale "worker=$LOAD_WORKER_REPLICAS" worker
} >>"$output_dir/worker-replacements.txt"
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
done
wait_for_cluster >"$output_dir/cluster-after-replacements.txt"
COMPOSE_PROJECT_NAME=$LOAD_PROJECT "$ROOT/scripts/verify-realtime-cluster.sh" compose \
>"$output_dir/pubsub-after-replacements.txt"
worker_target=$(service_ids worker | head -n 1)
assert_scope "$worker_target" worker
docker exec "$worker_target" /app/bin/who_need_help rpc \
"run_id = \"$LABEL\"
{:ok, job} =
%{
\"confirmation\" => \"isolated-local-resilience-probe\",
\"run_id\" => run_id
}
|> WhoNeedHelp.Workers.LocalRetryProbe.new()
|> Oban.insert()
IO.puts(job.id)" >"$output_dir/retry-job-insert.txt"
retry_job_inserted=true
retry_deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
while ((SECONDS < retry_deadline)); do
# The run label is restricted above and passed as a psql value.
# shellcheck disable=SC2016
"${compose[@]}" exec -T db sh -c \
'psql --no-psqlrc --tuples-only --no-align --set ON_ERROR_STOP=1 \
--set run_id="$1" --username "$POSTGRES_USER" --dbname "$POSTGRES_DB"' \
sh "$LABEL" >"$output_dir/retry-job.json" <<'SQL'
SELECT json_build_object(
'id', id,
'state', state,
'attempt', attempt,
'max_attempts', max_attempts,
'error_count', cardinality(errors),
'worker', worker
)
FROM oban_jobs
WHERE worker = 'WhoNeedHelp.Workers.LocalRetryProbe'
AND args->>'run_id' = :'run_id'
ORDER BY id DESC
LIMIT 1;
SQL
if jq -e '
.state == "completed" and
.attempt == 2 and
.max_attempts == 2 and
.error_count == 1 and
.worker == "WhoNeedHelp.Workers.LocalRetryProbe"
' "$output_dir/retry-job.json" >/dev/null 2>&1; then
break
fi
sleep 1
done
if ! jq -e '
.state == "completed" and
.attempt == 2 and
.max_attempts == 2 and
.error_count == 1 and
.worker == "WhoNeedHelp.Workers.LocalRetryProbe"
' "$output_dir/retry-job.json" >/dev/null; then
echo "The isolated Oban retry probe did not complete its second attempt." >&2
exit 1
fi
delete_retry_job "$output_dir/retry-job-cleanup.txt"
if [[ "$(tail -n 1 "$output_dir/retry-job-cleanup.txt")" != "1" ]]; then
echo "The exact resilience probe job was not removed." >&2
exit 1
fi
retry_job_inserted=false
stop_probe
probe_pid=
jq -s '{
samples: length,
failures: (map(select(.status != "200")) | length),
nodes: (map(.body | fromjson? | .node) | map(select(. != null)) | unique)
}' "$probe_log" >"$output_dir/readiness-summary.json"
if ! jq -e '.samples > 0 and .failures == 0 and (.nodes | length) >= 2' \
"$output_dir/readiness-summary.json" >/dev/null; then
echo "Readiness was unavailable or did not reach multiple web replicas." >&2
exit 1
fi
"${compose[@]}" ps -a >"$output_dir/compose-after.txt"
"${compose[@]}" logs --since "$run_started_at" proxy web worker \
>"$output_dir/application.log" 2>&1
trap - EXIT HUP INT TERM
printf 'Resilience evidence: %s\n' "$output_dir"