test: verify replica failure and rolling recovery

This commit is contained in:
SimpleTest 2026-07-19 06:41:06 +03:00
parent 909e16c969
commit 594065deb8
16 changed files with 1134 additions and 19 deletions

View File

@ -4,6 +4,7 @@
HTTP_PORT=0 HTTP_PORT=0
MAILPIT_PORT=0 MAILPIT_PORT=0
TRAEFIK_TRUSTED_IPS=127.0.0.1/32 TRAEFIK_TRUSTED_IPS=127.0.0.1/32
TRAEFIK_RETRY_ATTEMPTS=3
TRAEFIK_PROJECT_CONSTRAINT=GENERATED_UNIQUE_E2E_PROJECT TRAEFIK_PROJECT_CONSTRAINT=GENERATED_UNIQUE_E2E_PROJECT
TRAEFIK_APP_NAME=GENERATED_UNIQUE_E2E_ROUTER TRAEFIK_APP_NAME=GENERATED_UNIQUE_E2E_ROUTER
TRAEFIK_DOCKER_NETWORK=GENERATED_UNIQUE_E2E_NETWORK TRAEFIK_DOCKER_NETWORK=GENERATED_UNIQUE_E2E_NETWORK

View File

@ -5,6 +5,7 @@ MAILPIT_PORT=8027
# Comma-separated proxy IP/CIDR values whose X-Forwarded-* headers Traefik # Comma-separated proxy IP/CIDR values whose X-Forwarded-* headers Traefik
# accepts. Keep loopback locally; set the exact VPN proxy address for staging. # accepts. Keep loopback locally; set the exact VPN proxy address for staging.
TRAEFIK_TRUSTED_IPS=127.0.0.1/32 TRAEFIK_TRUSTED_IPS=127.0.0.1/32
TRAEFIK_RETRY_ATTEMPTS=3
# Docker-provider isolation and names. A second Compose project must use its # Docker-provider isolation and names. A second Compose project must use its
# own project constraint, router/service name, Docker network, and Host rule. # own project constraint, router/service name, Docker network, and Host rule.
TRAEFIK_PROJECT_CONSTRAINT=who_need_help TRAEFIK_PROJECT_CONSTRAINT=who_need_help

View File

@ -10,6 +10,7 @@ PHX_HOST=load.local
PHX_SCHEME=https PHX_SCHEME=https
PHX_URL_PORT=443 PHX_URL_PORT=443
TRAEFIK_TRUSTED_IPS=127.0.0.1/32 TRAEFIK_TRUSTED_IPS=127.0.0.1/32
TRAEFIK_RETRY_ATTEMPTS=3
TRAEFIK_PROJECT_CONSTRAINT=who_need_help_load TRAEFIK_PROJECT_CONSTRAINT=who_need_help_load
TRAEFIK_APP_NAME=who-need-help-load TRAEFIK_APP_NAME=who-need-help-load
TRAEFIK_DOCKER_NETWORK=who_need_help_load_internal TRAEFIK_DOCKER_NETWORK=who_need_help_load_internal
@ -55,3 +56,6 @@ LOAD_AUTH_VUS=8
LOAD_AUTH_WS_TIMEOUT_MS=5000 LOAD_AUTH_WS_TIMEOUT_MS=5000
LOAD_AUTH_THINK_SECONDS=0.1 LOAD_AUTH_THINK_SECONDS=0.1
LOAD_FIXTURE_PASSWORD=GENERATE_LOAD_FIXTURE_PASSWORD LOAD_FIXTURE_PASSWORD=GENERATE_LOAD_FIXTURE_PASSWORD
LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS=120
LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS=0.05
LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS=2

View File

@ -270,6 +270,23 @@ that dump. After a successful rollout it also removes the obsolete chart Secret
and only the local Helm history revisions that stored the former inline and only the local Helm history revisions that stored the former inline
credential fields. credential fields.
Exercise the verified local rolling-update path without recreating PostGIS or
the Secret:
```bash
./scripts/kind-rolling-verify.sh local-kind-rollout
```
Compose crash/replacement and Oban retry checks use the separate load project:
```bash
./scripts/load-stack-up.sh
./scripts/load-resilience-run.sh local-resilience
```
Both scripts retain ignored evidence under `output/resilience/`; their exact
mutation and cleanup boundaries are documented in the operations runbook.
For an external cluster, provide a real PostgreSQL/PostGIS service and a For an external cluster, provide a real PostgreSQL/PostGIS service and a
pre-created Secret through required `existingSecret`; the chart never renders pre-created Secret through required `existingSecret`; the chart never renders
credentials from tracked values. The Secret must contain `DATABASE_URL`, credentials from tracked values. The Secret must contain `DATABASE_URL`,

View File

@ -12,12 +12,12 @@ services:
web: web:
image: who-need-help:load image: who-need-help:load
labels: labels:
- traefik.http.routers.${TRAEFIK_APP_NAME}.middlewares=${TRAEFIK_APP_NAME}-forwarded - traefik.http.routers.${TRAEFIK_APP_NAME}.middlewares=${TRAEFIK_APP_NAME}-forwarded,${TRAEFIK_APP_NAME}-retry
- traefik.http.middlewares.${TRAEFIK_APP_NAME}-forwarded.headers.customrequestheaders.X-Forwarded-Proto=https - traefik.http.middlewares.${TRAEFIK_APP_NAME}-forwarded.headers.customrequestheaders.X-Forwarded-Proto=https
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.rule=${TRAEFIK_ROUTER_RULE} - traefik.http.routers.${TRAEFIK_APP_NAME}-tls.rule=${TRAEFIK_ROUTER_RULE}
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.entrypoints=websecure - traefik.http.routers.${TRAEFIK_APP_NAME}-tls.entrypoints=websecure
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.service=${TRAEFIK_APP_NAME} - traefik.http.routers.${TRAEFIK_APP_NAME}-tls.service=${TRAEFIK_APP_NAME}
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.middlewares=${TRAEFIK_APP_NAME}-forwarded - traefik.http.routers.${TRAEFIK_APP_NAME}-tls.middlewares=${TRAEFIK_APP_NAME}-forwarded,${TRAEFIK_APP_NAME}-retry
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.tls=true - traefik.http.routers.${TRAEFIK_APP_NAME}-tls.tls=true
worker: worker:

View File

@ -101,6 +101,8 @@ services:
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.rule=${TRAEFIK_ROUTER_RULE:-PathPrefix(`/`)} - traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.rule=${TRAEFIK_ROUTER_RULE:-PathPrefix(`/`)}
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.entrypoints=web - traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.entrypoints=web
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.service=${TRAEFIK_APP_NAME:-who-need-help} - traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.service=${TRAEFIK_APP_NAME:-who-need-help}
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.middlewares=${TRAEFIK_APP_NAME:-who-need-help}-retry
- traefik.http.middlewares.${TRAEFIK_APP_NAME:-who-need-help}-retry.retry.attempts=${TRAEFIK_RETRY_ATTEMPTS:-3}
- traefik.http.services.${TRAEFIK_APP_NAME:-who-need-help}.loadbalancer.server.port=4000 - traefik.http.services.${TRAEFIK_APP_NAME:-who-need-help}.loadbalancer.server.port=4000
healthcheck: healthcheck:
test: ["CMD", "curl", "--fail", "--silent", "http://localhost:4000/healthz/ready"] test: ["CMD", "curl", "--fail", "--silent", "http://localhost:4000/healthz/ready"]

View File

@ -10,13 +10,13 @@ item below unless the evidence column explicitly describes a local mock.
| Browser E2E | Manual headed-Chrome scenarios exist; no committed browser suite | A fresh uniquely named Compose project runs two-user urgent help, Activity, moderation, privacy, and error paths; traces are retained on failure; its exact volume is removed | | Browser E2E | Manual headed-Chrome scenarios exist; no committed browser suite | A fresh uniquely named Compose project runs two-user urgent help, Activity, moderation, privacy, and error paths; traces are retained on failure; its exact volume is removed |
| Android UI | Two JVM unit-test files; no `androidTest` source set | Emulator instrumentation covers deep links, permissions, foreground tracking, notification Stop, lifecycle, and network failure | | Android UI | Two JVM unit-test files; no `androidTest` source set | Emulator instrumentation covers deep links, permissions, foreground tracking, notification Stop, lifecycle, and network failure |
| CI and quality | No tracked CI workflow or static/security analysis dependencies | The same containerized gates pass locally and are represented in a validated CI workflow | | CI and quality | No tracked CI workflow or static/security analysis dependencies | The same containerized gates pass locally and are represented in a validated CI workflow |
| Localization and accessibility | Completed locally: product copy and custom validation messages are extracted; EN/UK/RU catalogs and localized category descriptions/structured values are implemented | 508 default and 40 error messages are current; RU/UK have no empty/fuzzy entries; 161 backend tests and all 8 browser specs pass, including locale persistence, keyboard, axe, themes, responsive widths, and reconnect | | Localization and accessibility | Completed locally: product copy and custom validation messages are extracted; EN/UK/RU catalogs and localized category descriptions/structured values are implemented | 508 default and 40 error messages are current; RU/UK have no empty/fuzzy entries; 163 backend tests and all 8 browser specs pass, including locale persistence, keyboard, axe, themes, responsive widths, and reconnect |
| Database scale | Core discovery/chat/moderation lists call unbounded `Repo.all()` | Cursor-bounded queries pass behavior tests and measured `EXPLAIN ANALYZE` checks on an isolated generated dataset | | Database scale | Core discovery/chat/moderation lists call unbounded `Repo.all()` | Cursor-bounded queries pass behavior tests and measured `EXPLAIN ANALYZE` checks on an isolated generated dataset |
| Load and resilience | Public/readiness/heartbeat k6 profile exists | Authenticated writes, chat, tracking, reconnect, rolling replacement, and worker retry profiles pass without touching staging data | | Load and resilience | Public/readiness/heartbeat k6 profile exists | Authenticated writes, chat, tracking, reconnect, rolling replacement, and worker retry profiles pass without touching staging data |
| Observability | Protected Prometheus text endpoint exists | Local Prometheus/Grafana/Alertmanager profile scrapes every replica and an induced isolated failure exercises alert delivery | | Observability | Protected Prometheus text endpoint exists | Local Prometheus/Grafana/Alertmanager profile scrapes every replica and an induced isolated failure exercises alert delivery |
| Backup | Validated local custom-format dump and restore drill exist | An encrypted artifact is uploaded to local S3-compatible MinIO and restored into a fresh database; corruption and interrupted-upload checks fail closed | | Backup | Validated local custom-format dump and restore drill exist | An encrypted artifact is uploaded to local S3-compatible MinIO and restored into a fresh database; corruption and interrupted-upload checks fail closed |
| External boundaries | Mailpit and a fake GitHub strategy cover parts of SMTP/OAuth | Local protocol-level SMTP/OAuth mocks and the applicable push adapter boundary cover success, rejection, retry, replay, and timeout | | External boundaries | Mailpit and a fake GitHub strategy cover parts of SMTP/OAuth | Local protocol-level SMTP/OAuth mocks and the applicable push adapter boundary cover success, rejection, retry, replay, and timeout |
| Final regression | 161 Phoenix tests plus reproducible browser and Android device suites | Browser, Android, API, DB, WebSocket, backup, monitoring, failure, cleanup, docs, and clean Git are verified from the final commits | | Final regression | 163 Phoenix tests plus reproducible browser and Android device suites | Browser, Android, API, DB, WebSocket, backup, monitoring, failure, cleanup, docs, and clean Git are verified from the final commits |
The goal remains open while any row lacks reproducible local evidence. The goal remains open while any row lacks reproducible local evidence.
@ -55,7 +55,7 @@ The goal remains open while any row lacks reproducible local evidence.
- The containerized `scripts/quality.sh` gate passes ShellCheck, Hadolint, - The containerized `scripts/quality.sh` gate passes ShellCheck, Hadolint,
actionlint, all Compose renders, Helm lint, a Trivy scan of tracked source and actionlint, all Compose renders, Helm lint, a Trivy scan of tracked source and
the rendered Kubernetes manifest, compiler/xref/Credo/Sobelow/Dialyzer/Hex the rendered Kubernetes manifest, compiler/xref/Credo/Sobelow/Dialyzer/Hex
checks, 161 Phoenix tests, both npm audits, and a Trivy scan of the production checks, 163 Phoenix tests, both npm audits, and a Trivy scan of the production
release image. It creates random one-run database credentials and removes its release image. It creates random one-run database credentials and removes its
exact volume, networks, images, and source snapshot. exact volume, networks, images, and source snapshot.
- The checked-in GitHub Actions workflow runs the same isolated backend/security - The checked-in GitHub Actions workflow runs the same isolated backend/security
@ -78,5 +78,16 @@ The goal remains open while any row lacks reproducible local evidence.
failure, retained no current positions after stop, passed cross-node PubSub failure, retained no current positions after stop, passed cross-node PubSub
and readiness, and restored every tracked table count after exact fixture and readiness, and restored every tracked table count after exact fixture
cleanup. cleanup.
- The same isolated profile now passes deliberate web/worker BEAM crashes,
sequential replacement of every replica, exact five-node cluster/PubSub
checks, and an Oban job that records one failure before succeeding on its
second attempt. A 743-sample readiness probe observed no final HTTP failure
and the probe job/domain fixtures were absent afterward.
- The project-owned kind cluster also passes a full web/worker rolling restart:
all four pod UIDs changed, all replacements became Ready with zero restarts,
the four-node cluster/PubSub probe passed, and database counts were unchanged.
The local single-node NodePort needed three reconnect attempts across 305
ultimately successful samples; this is recorded rather than presented as
raw transport continuity.
- The remaining rows above are still pending; this document is not a - The remaining rows above are still pending; this document is not a
completion claim for the entire hardening goal. completion claim for the entire hardening goal.

View File

@ -68,6 +68,52 @@ on one connected BEAM node and broadcasts from another.
Health checks do not replace alerting, database backups, restore drills, or Health checks do not replace alerting, database backups, restore drills, or
application-level synthetic checks. application-level synthetic checks.
## Local failure and rolling-replacement drills
The isolated load project can exercise process crashes, sequential container
replacement, and a real Oban retry without touching the normal Compose project:
```bash
./scripts/load-stack-up.sh
./scripts/load-resilience-run.sh local-resilience
```
The resilience script refuses `LOAD_PROJECT=who_need_help` and verifies the
Compose project/service labels of every container before stopping it. It:
1. continuously calls readiness through the isolated Traefik route;
2. terminates the BEAM process in one web and one worker container and requires
Docker's observed restart count to increase;
3. removes and replaces each web and worker replica one at a time;
4. waits for every configured BEAM node, then runs the cross-node PubSub probe;
5. enqueues a side-effect-free local worker that fails its first Oban attempt
and succeeds on its second;
6. removes that exact Oban row and requires no fixture domain rows to remain.
Traefik's retry middleware is attached to the HTTP and local TLS routers. Its
attempt count is an environment input. Traefik retries transport failures and,
with the checked configuration, does not opt in to retrying non-idempotent
requests. This reduces a stale-backend window; it is not a claim of production
availability.
For the project-owned kind cluster, run:
```bash
./scripts/kind-rolling-verify.sh local-kind-rollout
```
That script requires both the kind ownership marker and the control-plane
cluster label before invoking `rollout restart`. It changes only the web and
worker Deployment pod templates. It snapshots application-table counts before
and after, continuously probes the observed Docker mapping for the chart's
NodePort, requires all four old pod UIDs to disappear, waits for the exact BEAM
peer count, and verifies cross-node PubSub. PostGIS, its hostPath, the
Kubernetes Secret, and the namespace are not recreated.
The rollout timeout, probe interval/timeout/retry count, and cluster-join
timeout are experiment inputs. They are not production SLOs or resource
requirements.
## Protected Prometheus metrics ## Protected Prometheus metrics
The web role exposes Prometheus text format at `/metrics`. It requires the The web role exposes Prometheus text format at `/metrics`. It requires the

View File

@ -173,6 +173,37 @@ Ignored evidence:
- `output/performance/auth-final-20260719i/` - `output/performance/auth-final-20260719i/`
## Observed local resilience drills
The canonical Compose drill on 2026-07-19 used the isolated 3-web/2-worker
profile. One web and one worker BEAM process exited with status 1 and each
container's observed restart count increased to 1. All original replicas were
then replaced sequentially. The route returned 743 successful readiness
responses with zero final failure and responses from all three web nodes.
After replacement, the observed cluster contained all five BEAM nodes and the
cross-node PubSub probe passed.
The local Oban probe completed with state `completed`, attempt `2`,
`max_attempts=2`, and exactly one recorded first-attempt error. Its exact row
was removed afterward. The load database then contained zero probe jobs, users,
requests, messages, and tracking sessions. Run-scoped available logs contained
no unexpected application error, warning, HTTP 4xx/5xx, or database deadlock.
The canonical kind drill rolled both 2-replica Deployments from revision 17 to
18 with the chart's observed `maxUnavailable=0` and `maxSurge=1`. All four pod
UIDs changed, all replacements were Ready with zero container restart, the
four-node BEAM cluster and PubSub probe passed, and the application-table count
diff was empty. Of 305 readiness samples, all ultimately returned HTTP 200.
Two samples needed three transport retries in total while kind's single-node
NodePort reset connections during endpoint changes. Those retries are retained
in evidence rather than reported as uninterrupted raw TCP connections. This is
a local kind observation, not a production availability guarantee.
Ignored evidence:
- `output/resilience/compose-resilience-canonical-20260719f/`
- `output/resilience/kind-rollout-canonical-20260719c/`
Stop the isolated containers without deleting their database volume: Stop the isolated containers without deleting their database volume:
```sh ```sh

View File

@ -16,14 +16,14 @@ results from product limits and unknown production properties.
| Consent-driven live tracking | Implemented and cross-client verified | On API 37, Android started `TrackingService` as a location foreground service with a persistent Stop notification. After Home minimized the Activity, an emulator coordinate change reached PostGIS. Notification Stop removed the service, notification, active session, and raw position. | Browsers stop with the page. Android has no `ACCESS_BACKGROUND_LOCATION`, unattended start, or route history. | | Consent-driven live tracking | Implemented and cross-client verified | On API 37, Android started `TrackingService` as a location foreground service with a persistent Stop notification. After Home minimized the Activity, an emulator coordinate change reached PostGIS. Notification Stop removed the service, notification, active session, and raw position. | Browsers stop with the page. Android has no `ACCESS_BACKGROUND_LOCATION`, unattended start, or route history. |
| Privacy settings | Implemented and browser-verified | The profile exposed hidden, approximate public, exact for active match, and explicit exact-public options. Blocking and current-position cleanup have automated tests. | Exact public location remains a user opt-in; legal privacy and retention text still requires jurisdiction-specific review before launch. | | Privacy settings | Implemented and browser-verified | The profile exposed hidden, approximate public, exact for active match, and explicit exact-public options. Blocking and current-position cleanup have automated tests. | Exact public location remains a user opt-in; legal privacy and retention text still requires jurisdiction-specific review before launch. |
| Reputation and anti-abuse | Implemented at MVP level | Handover codes, two-party completion, double-blind reviews, unique-counterpart ranking, optional movement/proximity evidence, reports, blocks, abuse signals, and moderator audit paths have automated tests. | The system is not bot-proof and does not claim identity verification. No punitive numeric policy is enabled without measured and approved thresholds. | | Reputation and anti-abuse | Implemented at MVP level | Handover codes, two-party completion, double-blind reviews, unique-counterpart ranking, optional movement/proximity evidence, reports, blocks, abuse signals, and moderator audit paths have automated tests. | The system is not bot-proof and does not claim identity verification. No punitive numeric policy is enabled without measured and approved thresholds. |
| Social profiles | Manual links implemented; GitHub verification implemented and automated-tested | Manual links cannot set verification fields. The optional GitHub flow uses state, PKCE, a user-bound one-time session, unique provider ownership, and an audit record; 161 tests pass, including callback replay/state checks. No access-token field exists and the controller receives only normalized identity attributes. | The staging operator has not supplied GitHub OAuth credentials, so the real external provider redirect/callback remains disabled and has not been browser-verified. Other providers remain manual/unverified. | | Social profiles | Manual links implemented; GitHub verification implemented and automated-tested | Manual links cannot set verification fields. The optional GitHub flow uses state, PKCE, a user-bound one-time session, unique provider ownership, and an audit record; 163 tests pass, including callback replay/state checks. No access-token field exists and the controller receives only normalized identity attributes. | The staging operator has not supplied GitHub OAuth credentials, so the real external provider redirect/callback remains disabled and has not been browser-verified. Other providers remain manual/unverified. |
| Voluntary thanks | Implemented as an external optional link | A helper can expose an optional link after completion; the UI states that the platform does not process the payment. | The platform does not provide payments, escrow, refunds, tax reporting, or payment guarantees. | | Voluntary thanks | Implemented as an external optional link | A helper can expose an optional link after completion; the UI states that the platform does not process the payment. | The platform does not provide payments, escrow, refunds, tax reporting, or payment guarantees. |
| Android client | Local and public-staging clients implemented and emulator-verified | The native packages `org.whoneedhelp.mobile.debug` and `org.whoneedhelp.mobile.staging` launch the same authenticated LiveView app. Public HTTPS login, map, two-way chat, permission prompts, minimized foreground-service location updates, notification Stop, deep-link routing, and server cleanup were exercised on API 37. | Production signing, Play Store publication, verified Android App Links, unattended/background-permission tracking, and iOS are not implemented. | | Android client | Local and public-staging clients implemented and emulator-verified | The native packages `org.whoneedhelp.mobile.debug` and `org.whoneedhelp.mobile.staging` launch the same authenticated LiveView app. Public HTTPS login, map, two-way chat, permission prompts, minimized foreground-service location updates, notification Stop, deep-link routing, and server cleanup were exercised on API 37. | Production signing, Play Store publication, verified Android App Links, unattended/background-permission tracking, and iOS are not implemented. |
| Multiple web/worker instances | Implemented and locally verified | Docker Compose and kind each ran 2 web and 2 worker replicas. The project probes cross-node Phoenix PubSub using different BEAM nodes. Kubernetes web/worker pods were Ready with zero restarts at the final observation. | Local PostGIS is a single instance. Production database HA, backups, and recovery are operator work and are not claimed complete. | | Multiple web/worker instances | Implemented and locally failure/rollout-verified | The isolated Compose profile passed BEAM crashes and sequential replacement with 3 web/2 worker replicas, all five nodes joined, PubSub passed, and 743/743 readiness requests succeeded. The project-owned kind cluster replaced all 2 web/2 worker pod UIDs under `maxUnavailable=0`; all four replacement pods joined and PubSub passed. | Local PostGIS is a single instance. Production database HA, backups, and recovery are operator work and are not claimed complete. |
## Reproducible checks ## Reproducible checks
- The isolated Phoenix suite completed on 2026-07-19 with 161 - The isolated Phoenix suite completed on 2026-07-19 with 163
tests and 0 failures after cursor pagination, database aggregation, and the tests and 0 failures after cursor pagination, database aggregation, and the
full localization changes full localization changes
on Elixir 1.20.2 and Erlang/OTP 29.0.3. on Elixir 1.20.2 and Erlang/OTP 29.0.3.
@ -32,7 +32,7 @@ results from product limits and unknown production properties.
- `./scripts/quality.sh` passed ShellCheck 0.11.0, Hadolint 2.14.0 at warning - `./scripts/quality.sh` passed ShellCheck 0.11.0, Hadolint 2.14.0 at warning
threshold, actionlint 1.7.12, all four Compose renders, Helm lint, Trivy threshold, actionlint 1.7.12, all four Compose renders, Helm lint, Trivy
source/rendered-manifest scanning, xref, Credo high-priority checks, Sobelow source/rendered-manifest scanning, xref, Credo high-priority checks, Sobelow
strict/private checks, Hex audit, 161 Phoenix tests, both npm audits, and the strict/private checks, Hex audit, 163 Phoenix tests, both npm audits, and the
production-image vulnerability scan. The rendered Helm manifest and Debian production-image vulnerability scan. The rendered Helm manifest and Debian
13.6 release image each reported zero HIGH/CRITICAL findings under the 13.6 release image each reported zero HIGH/CRITICAL findings under the
configured gates. configured gates.
@ -67,6 +67,17 @@ results from product limits and unknown production properties.
containers also used read-only root filesystems and dropped capabilities. containers also used read-only root filesystems and dropped capabilities.
The cross-node probe observed all four BEAM nodes, and live/readiness returned The cross-node probe observed all four BEAM nodes, and live/readiness returned
`ok`/`ready`. Database counts remained `0 users / 0 help requests`. `ok`/`ready`. Database counts remained `0 users / 0 help requests`.
- The isolated Compose resilience drill observed a restart count increase for
one crashed web and worker BEAM process, replaced every replica sequentially,
observed all 5 cluster nodes, passed PubSub, and completed a real Oban retry
on attempt 2 after one recorded error. Its exact job row and all fixture
domain rows were absent afterward; 743 readiness samples had no final
failure.
- The reproducible kind rolling drill advanced both Deployments from revision
17 to 18, replaced all four pod UIDs, observed all 4 BEAM nodes, passed
PubSub, and left the database-count diff empty. All 305 readiness samples
ultimately returned 200; two samples needed three total reconnect attempts
during local single-node NodePort endpoint replacement.
- The committed browser suite passed its 1/1 bootstrap and all 8/8 Chromium - The committed browser suite passed its 1/1 bootstrap and all 8/8 Chromium
specs against a fresh PostGIS volume with two web and two worker replicas on specs against a fresh PostGIS volume with two web and two worker replicas on
2026-07-19. The retained successful-run artifact directory is 2026-07-19. The retained successful-run artifact directory is

View File

@ -0,0 +1,36 @@
defmodule WhoNeedHelp.Workers.LocalRetryProbe do
@moduledoc """
A side-effect-free worker used by the isolated local resilience profile.
The first execution returns an expected error and the second succeeds. No
application flow enqueues this worker; the resilience script requires an
explicit confirmation value and removes its exact Oban row after collecting
evidence.
"""
use Oban.Worker, queue: :maintenance, max_attempts: 2, tags: ["local-resilience-probe"]
@confirmation "isolated-local-resilience-probe"
@impl Oban.Worker
def perform(%Oban.Job{
args: %{"confirmation" => @confirmation, "run_id" => run_id},
attempt: 1
})
when is_binary(run_id) and run_id != "" do
{:error, :expected_first_attempt_failure}
end
def perform(%Oban.Job{
args: %{"confirmation" => @confirmation, "run_id" => run_id},
attempt: attempt
})
when is_binary(run_id) and run_id != "" and attempt >= 2 do
:ok
end
def perform(_job), do: {:cancel, :invalid_local_resilience_probe}
@impl Oban.Worker
def backoff(_job), do: 1
end

View File

@ -7,7 +7,15 @@ target="$ROOT/.env.e2e"
if [ -f "$target" ]; then if [ -f "$target" ]; then
chmod 600 "$target" chmod 600 "$target"
echo ".env.e2e already exists; no secret was changed."
if ! grep -q '^TRAEFIK_RETRY_ATTEMPTS=' "$target"; then
printf '\nTRAEFIK_RETRY_ATTEMPTS=3\n' >>"$target"
chmod 600 "$target"
echo "Added the missing Traefik retry input; no E2E secret was changed."
else
echo ".env.e2e already exists; no secret or experiment input was changed."
fi
exit 0 exit 0
fi fi
@ -21,6 +29,7 @@ cat >"$target" <<EOF
HTTP_PORT=0 HTTP_PORT=0
MAILPIT_PORT=0 MAILPIT_PORT=0
TRAEFIK_TRUSTED_IPS=127.0.0.1/32 TRAEFIK_TRUSTED_IPS=127.0.0.1/32
TRAEFIK_RETRY_ATTEMPTS=3
TRAEFIK_PROJECT_CONSTRAINT=generated-per-run TRAEFIK_PROJECT_CONSTRAINT=generated-per-run
TRAEFIK_APP_NAME=generated-per-run TRAEFIK_APP_NAME=generated-per-run
TRAEFIK_DOCKER_NETWORK=generated-per-run TRAEFIK_DOCKER_NETWORK=generated-per-run

View File

@ -15,23 +15,72 @@ done
if [ -f "$ENV_FILE" ]; then if [ -f "$ENV_FILE" ]; then
chmod 600 "$ENV_FILE" chmod 600 "$ENV_FILE"
if grep -q '^LOAD_FIXTURE_PASSWORD=' "$ENV_FILE"; then needs_fixture_password=true
needs_resilience_timeout=true
needs_resilience_interval=true
needs_resilience_request_timeout=true
needs_traefik_retry_attempts=true
grep -q '^LOAD_FIXTURE_PASSWORD=' "$ENV_FILE" && needs_fixture_password=false
grep -q '^LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS=' "$ENV_FILE" &&
needs_resilience_timeout=false
grep -q '^LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS=' "$ENV_FILE" &&
needs_resilience_interval=false
grep -q '^LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS=' "$ENV_FILE" &&
needs_resilience_request_timeout=false
grep -q '^TRAEFIK_RETRY_ATTEMPTS=' "$ENV_FILE" && needs_traefik_retry_attempts=false
if [ "$needs_fixture_password" = false ] &&
[ "$needs_resilience_timeout" = false ] &&
[ "$needs_resilience_interval" = false ] &&
[ "$needs_resilience_request_timeout" = false ] &&
[ "$needs_traefik_retry_attempts" = false ]; then
echo ".env.load already exists; no secret or experiment input was changed." echo ".env.load already exists; no secret or experiment input was changed."
exit 0 exit 0
fi fi
umask 077 umask 077
load_fixture_password=$(openssl rand -hex 24) load_fixture_password=
if [ "$needs_fixture_password" = true ]; then
load_fixture_password=$(openssl rand -hex 24)
fi
{ {
printf '\n# Added by the authenticated-load profile upgrade.\n' if [ "$needs_fixture_password" = true ]; then
printf 'LOAD_AUTH_VUS=8\n' printf '\n# Added by the authenticated-load profile upgrade.\n'
printf 'LOAD_AUTH_WS_TIMEOUT_MS=5000\n' printf 'LOAD_AUTH_VUS=8\n'
printf 'LOAD_AUTH_THINK_SECONDS=0.1\n' printf 'LOAD_AUTH_WS_TIMEOUT_MS=5000\n'
printf 'LOAD_FIXTURE_PASSWORD=%s\n' "$load_fixture_password" printf 'LOAD_AUTH_THINK_SECONDS=0.1\n'
printf 'LOAD_FIXTURE_PASSWORD=%s\n' "$load_fixture_password"
fi
if [ "$needs_resilience_timeout" = true ] ||
[ "$needs_resilience_interval" = true ]; then
printf '\n# Added by the local resilience-profile upgrade.\n'
fi
if [ "$needs_resilience_timeout" = true ]; then
printf 'LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS=120\n'
fi
if [ "$needs_resilience_interval" = true ]; then
printf 'LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS=0.05\n'
fi
if [ "$needs_resilience_request_timeout" = true ]; then
printf 'LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS=2\n'
fi
if [ "$needs_traefik_retry_attempts" = true ]; then
printf 'TRAEFIK_RETRY_ATTEMPTS=3\n'
fi
} >>"$ENV_FILE" } >>"$ENV_FILE"
chmod 600 "$ENV_FILE" chmod 600 "$ENV_FILE"
unset load_fixture_password unset load_fixture_password needs_fixture_password needs_resilience_timeout \
echo "Added authenticated-load inputs and a random fixture password to ignored .env.load." needs_resilience_interval needs_resilience_request_timeout \
needs_traefik_retry_attempts
echo "Added missing authenticated-load/resilience inputs to ignored .env.load."
exit 0 exit 0
fi fi

417
scripts/kind-rolling-verify.sh Executable file
View File

@ -0,0 +1,417 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)
KUBECTL="$ROOT/.tools/bin/kubectl"
CONTEXT=kind-who-need-help
CLUSTER_CONTAINER=who-need-help-control-plane
NAMESPACE=who-need-help
WEB_DEPLOYMENT=who-need-help-who-need-help-web
WORKER_DEPLOYMENT=who-need-help-who-need-help-worker
SERVICE=who-need-help-who-need-help
OWNERSHIP_MARKER="$ROOT/.tools/who-need-help.owned"
ROLLOUT_TIMEOUT=${KIND_ROLLOUT_TIMEOUT:-180s}
PROBE_INTERVAL=${KIND_ROLLOUT_PROBE_INTERVAL_SECONDS:-0.05}
PROBE_TIMEOUT=${KIND_ROLLOUT_PROBE_TIMEOUT_SECONDS:-2}
PROBE_RETRIES=${KIND_ROLLOUT_PROBE_RETRIES:-2}
CLUSTER_JOIN_TIMEOUT=${KIND_CLUSTER_JOIN_TIMEOUT_SECONDS:-120}
LABEL=${1:-"kind-rollout-$(date -u +%Y%m%dT%H%M%SZ)"}
if [[ ! -x "$KUBECTL" ]]; then
echo "The project-owned kubectl is missing. Run scripts/bootstrap-kubernetes-tools.sh." >&2
exit 1
fi
if [[ ! -f "$OWNERSHIP_MARKER" ]]; then
echo "The kind ownership marker is missing; refusing to mutate the cluster." >&2
exit 1
fi
if [[ ! "$LABEL" =~ ^[A-Za-z0-9._-]+$ ]]; then
echo "Run label may contain only letters, numbers, dot, underscore, and dash." >&2
exit 1
fi
if [[ ! "$ROLLOUT_TIMEOUT" =~ ^[1-9][0-9]*[smh]$ ]]; then
echo "KIND_ROLLOUT_TIMEOUT must be a positive Kubernetes duration in s, m, or h." >&2
exit 1
fi
if ! awk -v value="$PROBE_INTERVAL" \
'BEGIN {exit !(value ~ /^[0-9]+([.][0-9]+)?$/ && value > 0)}'; then
echo "KIND_ROLLOUT_PROBE_INTERVAL_SECONDS must be greater than zero." >&2
exit 1
fi
if [[ ! "$PROBE_TIMEOUT" =~ ^[1-9][0-9]*$ ]]; then
echo "KIND_ROLLOUT_PROBE_TIMEOUT_SECONDS must be a positive integer." >&2
exit 1
fi
if [[ ! "$PROBE_RETRIES" =~ ^[0-9]+$ ]]; then
echo "KIND_ROLLOUT_PROBE_RETRIES must be a non-negative integer." >&2
exit 1
fi
if [[ ! "$CLUSTER_JOIN_TIMEOUT" =~ ^[1-9][0-9]*$ ]]; then
echo "KIND_CLUSTER_JOIN_TIMEOUT_SECONDS must be a positive integer." >&2
exit 1
fi
kube=("$KUBECTL" --context "$CONTEXT" --namespace "$NAMESPACE")
if ! "$KUBECTL" config get-contexts -o name | grep -Fxq "$CONTEXT"; then
echo "The expected local kind context does not exist." >&2
exit 1
fi
if [[ "$(docker inspect --format '{{index .Config.Labels "io.x-k8s.kind.cluster"}}' \
"$CLUSTER_CONTAINER")" != "who-need-help" ]]; then
echo "The kind control-plane container does not belong to this project cluster." >&2
exit 1
fi
output_dir="$ROOT/output/resilience/$LABEL"
mkdir -p "$output_dir"
chmod 700 "$ROOT/output" "$ROOT/output/resilience" "$output_dir"
probe_marker="$output_dir/.probe-running"
probe_log="$output_dir/readiness.jsonl"
run_started_at=$(date -u +%Y-%m-%dT%H:%M:%SZ)
touch "$probe_marker"
deployment_snapshot() {
"${kube[@]}" get deployment "$WEB_DEPLOYMENT" "$WORKER_DEPLOYMENT" -o json |
jq '[
.items[] | {
name: .metadata.name,
generation: .metadata.generation,
revision: .metadata.annotations["deployment.kubernetes.io/revision"],
replicas: .spec.replicas,
ready: .status.readyReplicas,
available: .status.availableReplicas,
updated: .status.updatedReplicas,
strategy: .spec.strategy,
image: .spec.template.spec.containers[0].image
}
]' >"$1"
}
database_snapshot() {
# Variables are intentionally expanded inside the PostGIS container.
# shellcheck disable=SC2016
"${kube[@]}" exec -i postgis-0 -- sh -c \
'psql --no-psqlrc --tuples-only --no-align --set ON_ERROR_STOP=1 \
--username "$POSTGRES_USER" --dbname "$POSTGRES_DB"' >"$1" <<'SQL'
BEGIN READ ONLY;
SELECT 'users' AS table_name, count(*) AS row_count FROM users
UNION ALL SELECT 'users_tokens', count(*) FROM users_tokens
UNION ALL SELECT 'help_requests', count(*) FROM help_requests
UNION ALL SELECT 'messages', count(*) FROM messages
UNION ALL SELECT 'categories', count(*) FROM categories
UNION ALL SELECT 'help_assignments', count(*) FROM help_assignments
UNION ALL SELECT 'activities', count(*) FROM activities
UNION ALL SELECT 'reports', count(*) FROM reports
UNION ALL SELECT 'social_identities', count(*) FROM social_identities
UNION ALL SELECT 'tracking_sessions', count(*) FROM tracking_sessions
UNION ALL SELECT 'tracking_positions', count(*) FROM tracking_positions
UNION ALL SELECT 'schema_migrations', count(*) FROM schema_migrations
ORDER BY table_name;
COMMIT;
SQL
}
pod_snapshot() {
"${kube[@]}" get pods \
-l app.kubernetes.io/instance=who-need-help \
-o json |
jq '[
.items[]
| select(
.metadata.labels["app.kubernetes.io/component"] == "web" or
.metadata.labels["app.kubernetes.io/component"] == "worker"
)
| {
name: .metadata.name,
uid: .metadata.uid,
component: .metadata.labels["app.kubernetes.io/component"],
ready: ([.status.containerStatuses[]?.ready] | all),
restarts: ([.status.containerStatuses[]?.restartCount] | add // 0)
}
] | sort_by(.component, .name)' >"$1"
}
stop_probe() {
unlink "$probe_marker" 2>/dev/null || true
if [[ -n "${probe_pid:-}" ]]; then
wait "$probe_pid" 2>/dev/null || true
fi
}
cleanup() {
local status=$?
trap - EXIT HUP INT TERM
stop_probe
exit "$status"
}
trap cleanup EXIT HUP INT TERM
deployment_snapshot "$output_dir/deployments-before.json"
if ! jq -e '
length == 2 and
all(
.replicas >= 2 and
.ready == .replicas and
.available == .replicas and
.updated == .replicas and
.strategy.type == "RollingUpdate" and
.strategy.rollingUpdate.maxUnavailable == 0 and
.strategy.rollingUpdate.maxSurge == 1
)
' "$output_dir/deployments-before.json" >/dev/null; then
echo "The local deployments are not ready for the recorded rolling strategy." >&2
exit 1
fi
database_snapshot "$output_dir/database-before.txt"
pod_snapshot "$output_dir/pods-before.json"
node_port=$(
"${kube[@]}" get service "$SERVICE" \
-o jsonpath='{.spec.ports[?(@.name=="http")].nodePort}'
)
published=$(
docker port "$CLUSTER_CONTAINER" "${node_port}/tcp" |
head -n 1
)
host_port=${published##*:}
if [[ ! "$host_port" =~ ^[1-9][0-9]*$ ]]; then
echo "The kind HTTP NodePort has no observed Docker host mapping." >&2
exit 1
fi
base_url="http://127.0.0.1:$host_port"
sample_readiness() {
while [[ -e "$probe_marker" ]]; do
observed_at=$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ)
body_file="$output_dir/kind-readiness-body.$$"
error_file="$output_dir/kind-readiness-error.$$"
set +e
result=$(
curl --silent --show-error \
--max-time "$PROBE_TIMEOUT" \
--retry "$PROBE_RETRIES" \
--retry-all-errors \
--retry-connrefused \
--retry-delay 0 \
--output "$body_file" \
--write-out '%{http_code} %{num_retries}' \
"$base_url/healthz/ready" 2>"$error_file"
)
curl_status=$?
set -e
read -r status retries <<<"$result"
if [[ "$curl_status" -ne 0 ]]; then
status=000
fi
if [[ -f "$body_file" ]]; then
body=$(tr -d '\n' <"$body_file")
else
body=
fi
error=$(tr -d '\n' <"$error_file")
unlink "$body_file" 2>/dev/null || true
unlink "$error_file" 2>/dev/null || true
jq -cn \
--arg observed_at "$observed_at" \
--arg status "$status" \
--arg body "$body" \
--arg error "$error" \
--argjson retries "${retries:-0}" \
'{
observed_at: $observed_at,
status: $status,
retries: $retries,
body: $body,
error: $error
}' >>"$probe_log"
sleep "$PROBE_INTERVAL"
done
}
{
printf 'observed_at=%s\n' "$run_started_at"
printf 'context=%s\n' "$CONTEXT"
printf 'namespace=%s\n' "$NAMESPACE"
printf 'rollout_timeout=%s\n' "$ROLLOUT_TIMEOUT"
printf 'probe_interval_seconds=%s\n' "$PROBE_INTERVAL"
printf 'probe_timeout_seconds=%s\n' "$PROBE_TIMEOUT"
printf 'probe_retries=%s\n' "$PROBE_RETRIES"
printf 'cluster_join_timeout_seconds=%s\n' "$CLUSTER_JOIN_TIMEOUT"
printf 'node_port=%s\n' "$node_port"
printf 'observed_host_port=%s\n' "$host_port"
"$KUBECTL" version --client
} >"$output_dir/environment.txt"
sample_readiness &
probe_pid=$!
"${kube[@]}" rollout restart "deployment/$WEB_DEPLOYMENT" \
>"$output_dir/web-rollout-restart.txt"
"${kube[@]}" rollout status "deployment/$WEB_DEPLOYMENT" \
"--timeout=$ROLLOUT_TIMEOUT" >"$output_dir/web-rollout-status.txt"
"${kube[@]}" rollout restart "deployment/$WORKER_DEPLOYMENT" \
>"$output_dir/worker-rollout-restart.txt"
"${kube[@]}" rollout status "deployment/$WORKER_DEPLOYMENT" \
"--timeout=$ROLLOUT_TIMEOUT" >"$output_dir/worker-rollout-status.txt"
"${kube[@]}" wait \
--for=condition=Ready \
--timeout="$ROLLOUT_TIMEOUT" \
pod \
-l app.kubernetes.io/instance=who-need-help \
>"$output_dir/pods-ready.txt"
old_pod_deadline=$((SECONDS + CLUSTER_JOIN_TIMEOUT))
while ((SECONDS < old_pod_deadline)); do
"${kube[@]}" get pods \
-l app.kubernetes.io/instance=who-need-help \
-o json >"$output_dir/pods-current.json"
retained_count=$(
jq \
--slurpfile before "$output_dir/pods-before.json" \
'[
.items[].metadata.uid as $uid
| select($before[0] | any(.uid == $uid))
] | length' "$output_dir/pods-current.json"
)
if [[ "$retained_count" == "0" ]]; then
break
fi
sleep 1
done
unlink "$output_dir/pods-current.json" 2>/dev/null || true
if [[ "$retained_count" != "0" ]]; then
echo "Old application pods did not terminate after the rollout." >&2
exit 1
fi
deployment_snapshot "$output_dir/deployments-after.json"
pod_snapshot "$output_dir/pods-after.json"
database_snapshot "$output_dir/database-after.txt"
if ! diff -u "$output_dir/database-before.txt" "$output_dir/database-after.txt" \
>"$output_dir/database-diff.txt"; then
echo "The rolling verification changed tracked database counts." >&2
exit 1
fi
jq -n \
--slurpfile before "$output_dir/pods-before.json" \
--slurpfile after "$output_dir/pods-after.json" \
'{
before_count: ($before[0] | length),
after_count: ($after[0] | length),
retained_uids: (
[$before[0][].uid] as $old
| [$after[0][].uid | select(. as $uid | $old | index($uid))]
),
after_all_ready: ($after[0] | all(.ready and .restarts == 0))
}' >"$output_dir/pod-replacement-summary.json"
if ! jq -e '
.before_count == 4 and
.after_count == 4 and
.retained_uids == [] and
.after_all_ready
' "$output_dir/pod-replacement-summary.json" >/dev/null; then
echo "The rollout did not replace all four application pods cleanly." >&2
exit 1
fi
web_pod=$(
"${kube[@]}" get pod \
-l app.kubernetes.io/component=web \
--field-selector=status.phase=Running \
-o jsonpath='{.items[0].metadata.name}'
)
expected_peers=$(
jq '[.[].replicas] | add - 1' "$output_dir/deployments-after.json"
)
cluster_deadline=$((SECONDS + CLUSTER_JOIN_TIMEOUT))
while ((SECONDS < cluster_deadline)); do
peer_count=$(
"${kube[@]}" exec "$web_pod" -- /app/bin/who_need_help rpc \
'IO.puts(length(Node.list()))' 2>/dev/null |
tail -n 1
)
if [[ "$peer_count" == "$expected_peers" ]]; then
break
fi
sleep 1
done
if [[ "$peer_count" != "$expected_peers" ]]; then
echo "All rolled application pods did not join the BEAM cluster." >&2
exit 1
fi
"${kube[@]}" exec "$web_pod" -- /app/bin/who_need_help rpc \
'IO.inspect(%{node: node(), peers: Node.list(), peer_count: length(Node.list())})' \
>"$output_dir/cluster-after.txt"
"$ROOT/scripts/verify-realtime-cluster.sh" kind \
>"$output_dir/pubsub-after.txt"
stop_probe
probe_pid=
jq -s '{
samples: length,
failures: (map(select(.status != "200")) | length),
retried_samples: (map(select(.retries > 0)) | length),
total_retries: (map(.retries) | add),
nodes: (map(.body | fromjson? | .node) | map(select(. != null)) | unique)
}' "$probe_log" >"$output_dir/readiness-summary.json"
"${kube[@]}" get pods \
-l app.kubernetes.io/component=web \
-o json |
jq '[.items[].status.podIP | "who_need_help@" + .] | sort' \
>"$output_dir/current-web-nodes.json"
if ! jq -e \
--slurpfile current "$output_dir/current-web-nodes.json" '
.samples > 0 and
.failures == 0 and
($current[0] | length) == 2 and
(($current[0] - .nodes) | length) == 0
' "$output_dir/readiness-summary.json" >/dev/null; then
echo "Kind readiness was unavailable or did not reach both web replicas." >&2
exit 1
fi
"${kube[@]}" logs \
--selector app.kubernetes.io/instance=who-need-help \
--all-containers \
--since-time "$run_started_at" \
--prefix >"$output_dir/application.log" 2>&1
trap - EXIT HUP INT TERM
printf 'Kind rollout evidence: %s\n' "$output_dir"

456
scripts/load-resilience-run.sh Executable file
View File

@ -0,0 +1,456 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)
ENV_FILE="$ROOT/.env.load"
LABEL=${1:-"resilience-$(date -u +%Y%m%dT%H%M%SZ)"}
if [[ ! -f "$ENV_FILE" ]]; then
echo "Missing $ENV_FILE. Run scripts/ensure-local-load-env.sh first." >&2
exit 1
fi
set -a
# shellcheck source=/dev/null
. "$ENV_FILE"
set +a
for name in LOAD_PROJECT LOAD_HOST LOAD_WEB_REPLICAS LOAD_WORKER_REPLICAS \
LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS \
LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS \
LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS TRAEFIK_RETRY_ATTEMPTS \
HTTP_PORT POSTGRES_DB; do
if [[ -z "${!name:-}" ]]; then
echo "$name is missing from .env.load" >&2
exit 1
fi
done
if [[ "$LOAD_PROJECT" == "who_need_help" ]]; then
echo "The resilience profile must not use the staging Compose project." >&2
exit 1
fi
if [[ ! "$LABEL" =~ ^[A-Za-z0-9._-]+$ ]]; then
echo "Run label may contain only letters, numbers, dot, underscore, and dash." >&2
exit 1
fi
for name in LOAD_WEB_REPLICAS LOAD_WORKER_REPLICAS \
LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS \
LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS TRAEFIK_RETRY_ATTEMPTS; do
if [[ ! "${!name}" =~ ^[1-9][0-9]*$ ]]; then
echo "$name must be a positive integer." >&2
exit 1
fi
done
if ! awk -v value="$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS" \
'BEGIN {exit !(value ~ /^[0-9]+([.][0-9]+)?$/ && value > 0)}'; then
echo "LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS must be greater than zero." >&2
exit 1
fi
compose=(
docker compose
--env-file "$ENV_FILE"
-p "$LOAD_PROJECT"
-f compose.yaml
-f compose.load.yaml
)
output_dir="$ROOT/output/resilience/$LABEL"
mkdir -p "$output_dir"
chmod 700 "$ROOT/output" "$ROOT/output/resilience" "$output_dir"
run_started_at=$(date -u +%Y-%m-%dT%H:%M:%SZ)
probe_marker="$output_dir/.probe-running"
probe_log="$output_dir/readiness.jsonl"
touch "$probe_marker"
assert_scope() {
local container_id=$1
local expected_service=$2
local observed_project observed_service
observed_project=$(
docker inspect --format '{{index .Config.Labels "com.docker.compose.project"}}' \
"$container_id"
)
observed_service=$(
docker inspect --format '{{index .Config.Labels "com.docker.compose.service"}}' \
"$container_id"
)
if [[ "$observed_project" != "$LOAD_PROJECT" ||
"$observed_service" != "$expected_service" ]]; then
echo "Container scope mismatch for $container_id." >&2
exit 1
fi
}
service_ids() {
"${compose[@]}" ps -q "$1"
}
wait_for_service() {
local service=$1
local expected=$2
local require_health=$3
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
while ((SECONDS < deadline)); do
mapfile -t ids < <(service_ids "$service")
if [[ "${#ids[@]}" -eq "$expected" ]]; then
local ready=true
local id state health
for id in "${ids[@]}"; do
state=$(docker inspect --format '{{.State.Status}}' "$id")
health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{end}}' "$id")
if [[ "$state" != "running" ||
("$require_health" == "true" && "$health" != "healthy") ]]; then
ready=false
break
fi
done
if [[ "$ready" == "true" ]]; then
return 0
fi
fi
sleep 1
done
echo "Timed out waiting for $expected ready $service replicas." >&2
return 1
}
wait_for_restart() {
local container_id=$1
local previous_restarts=$2
local require_health=$3
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
while ((SECONDS < deadline)); do
local state restarts health
state=$(docker inspect --format '{{.State.Status}}' "$container_id")
restarts=$(docker inspect --format '{{.RestartCount}}' "$container_id")
health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{end}}' "$container_id")
if [[ "$state" == "running" && "$restarts" -gt "$previous_restarts" &&
("$require_health" == "false" || "$health" == "healthy") ]]; then
return 0
fi
sleep 1
done
echo "Timed out waiting for container $container_id to restart." >&2
return 1
}
wait_for_cluster() {
local expected_peers=$((LOAD_WEB_REPLICAS + LOAD_WORKER_REPLICAS - 1))
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
while ((SECONDS < deadline)); do
local target peer_count
target=$(service_ids web | head -n 1)
assert_scope "$target" web
peer_count=$(
docker exec "$target" /app/bin/who_need_help rpc \
'IO.puts(length(Node.list()))' 2>/dev/null |
tail -n 1
)
if [[ "$peer_count" == "$expected_peers" ]]; then
docker exec "$target" /app/bin/who_need_help rpc \
'IO.inspect(%{node: node(), peers: Node.list(), peer_count: length(Node.list())})'
return 0
fi
sleep 1
done
echo "Timed out waiting for all local BEAM replicas to join the cluster." >&2
return 1
}
sample_readiness() {
while [[ -e "$probe_marker" ]]; do
observed_at=$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ)
body_file="$output_dir/readiness-body.$$"
error_file="$output_dir/readiness-error.$$"
set +e
status=$(
curl --silent --show-error \
--max-time "$LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS" \
--header "Host: $LOAD_HOST" \
--output "$body_file" \
--write-out '%{http_code}' \
"http://localhost:$HTTP_PORT/healthz/ready" 2>"$error_file"
)
curl_status=$?
set -e
if [[ "$curl_status" -ne 0 ]]; then
status=000
fi
if [[ -f "$body_file" ]]; then
body=$(tr -d '\n' <"$body_file")
else
body=
fi
error=$(tr -d '\n' <"$error_file")
unlink "$body_file" 2>/dev/null || true
unlink "$error_file" 2>/dev/null || true
jq -cn \
--arg observed_at "$observed_at" \
--arg status "$status" \
--arg body "$body" \
--arg error "$error" \
'{observed_at: $observed_at, status: $status, body: $body, error: $error}' \
>>"$probe_log"
sleep "$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS"
done
}
stop_probe() {
unlink "$probe_marker" 2>/dev/null || true
if [[ -n "${probe_pid:-}" ]]; then
wait "$probe_pid" 2>/dev/null || true
fi
}
delete_retry_job() {
local destination=$1
local target
target=$(service_ids worker | head -n 1)
assert_scope "$target" worker
docker exec "$target" /app/bin/who_need_help rpc \
"import Ecto.Query
run_id = \"$LABEL\"
{count, _} =
WhoNeedHelp.Repo.delete_all(
from job in Oban.Job,
where:
job.worker == \"WhoNeedHelp.Workers.LocalRetryProbe\" and
fragment(\"?->>'run_id'\", job.args) == ^run_id
)
IO.puts(count)" >"$destination"
}
restore_topology() {
"${compose[@]}" up -d --no-deps \
--scale "web=$LOAD_WEB_REPLICAS" \
--scale "worker=$LOAD_WORKER_REPLICAS" \
web worker >/dev/null 2>&1 || true
}
cleanup() {
local status=$?
local cleanup_status=0
trap - EXIT HUP INT TERM
stop_probe
set +e
restore_topology
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
if [[ "${retry_job_inserted:-false}" == "true" ]]; then
delete_retry_job "$output_dir/retry-job-cleanup-on-exit.txt"
cleanup_status=$?
fi
set -e
if [[ "$cleanup_status" -ne 0 ]]; then
echo "Resilience probe cleanup failed." >&2
status=1
fi
exit "$status"
}
trap cleanup EXIT HUP INT TERM
wait_for_service web "$LOAD_WEB_REPLICAS" true
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
{
printf 'observed_at=%s\n' "$run_started_at"
printf 'load_project=%s\n' "$LOAD_PROJECT"
printf 'web_replicas=%s\n' "$LOAD_WEB_REPLICAS"
printf 'worker_replicas=%s\n' "$LOAD_WORKER_REPLICAS"
printf 'recovery_timeout_seconds=%s\n' "$LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS"
printf 'probe_interval_seconds=%s\n' "$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS"
printf 'request_timeout_seconds=%s\n' "$LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS"
printf 'traefik_retry_attempts=%s\n' "$TRAEFIK_RETRY_ATTEMPTS"
docker compose version
docker info --format 'docker_server={{.ServerVersion}}'
} >"$output_dir/environment.txt"
sample_readiness &
probe_pid=$!
web_target=$(service_ids web | head -n 1)
assert_scope "$web_target" web
web_restarts=$(docker inspect --format '{{.RestartCount}}' "$web_target")
docker exec "$web_target" /app/bin/who_need_help rpc ':erlang.halt(1)' \
>"$output_dir/web-kill.txt" 2>&1 || true
wait_for_restart "$web_target" "$web_restarts" true
docker inspect "$web_target" |
jq '.[0] | {
id: .Id,
project: .Config.Labels["com.docker.compose.project"],
service: .Config.Labels["com.docker.compose.service"],
restart_count: .RestartCount,
state: .State.Status,
health: .State.Health.Status
}' >"$output_dir/web-restart.json"
worker_target=$(service_ids worker | head -n 1)
assert_scope "$worker_target" worker
worker_restarts=$(docker inspect --format '{{.RestartCount}}' "$worker_target")
docker exec "$worker_target" /app/bin/who_need_help rpc ':erlang.halt(1)' \
>"$output_dir/worker-kill.txt" 2>&1 || true
wait_for_restart "$worker_target" "$worker_restarts" false
docker inspect "$worker_target" |
jq '.[0] | {
id: .Id,
project: .Config.Labels["com.docker.compose.project"],
service: .Config.Labels["com.docker.compose.service"],
restart_count: .RestartCount,
state: .State.Status
}' >"$output_dir/worker-restart.json"
mapfile -t original_web_ids < <(service_ids web)
for container_id in "${original_web_ids[@]}"; do
assert_scope "$container_id" web
{
docker stop --time 30 "$container_id"
docker rm "$container_id"
"${compose[@]}" up -d --no-deps --scale "web=$LOAD_WEB_REPLICAS" web
} >>"$output_dir/web-replacements.txt"
wait_for_service web "$LOAD_WEB_REPLICAS" true
done
mapfile -t original_worker_ids < <(service_ids worker)
for container_id in "${original_worker_ids[@]}"; do
assert_scope "$container_id" worker
{
docker stop --time 30 "$container_id"
docker rm "$container_id"
"${compose[@]}" up -d --no-deps --scale "worker=$LOAD_WORKER_REPLICAS" worker
} >>"$output_dir/worker-replacements.txt"
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
done
wait_for_cluster >"$output_dir/cluster-after-replacements.txt"
COMPOSE_PROJECT_NAME=$LOAD_PROJECT "$ROOT/scripts/verify-realtime-cluster.sh" compose \
>"$output_dir/pubsub-after-replacements.txt"
worker_target=$(service_ids worker | head -n 1)
assert_scope "$worker_target" worker
docker exec "$worker_target" /app/bin/who_need_help rpc \
"run_id = \"$LABEL\"
{:ok, job} =
%{
\"confirmation\" => \"isolated-local-resilience-probe\",
\"run_id\" => run_id
}
|> WhoNeedHelp.Workers.LocalRetryProbe.new()
|> Oban.insert()
IO.puts(job.id)" >"$output_dir/retry-job-insert.txt"
retry_job_inserted=true
retry_deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
while ((SECONDS < retry_deadline)); do
# The run label is restricted above and passed as a psql value.
# shellcheck disable=SC2016
"${compose[@]}" exec -T db sh -c \
'psql --no-psqlrc --tuples-only --no-align --set ON_ERROR_STOP=1 \
--set run_id="$1" --username "$POSTGRES_USER" --dbname "$POSTGRES_DB"' \
sh "$LABEL" >"$output_dir/retry-job.json" <<'SQL'
SELECT json_build_object(
'id', id,
'state', state,
'attempt', attempt,
'max_attempts', max_attempts,
'error_count', cardinality(errors),
'worker', worker
)
FROM oban_jobs
WHERE worker = 'WhoNeedHelp.Workers.LocalRetryProbe'
AND args->>'run_id' = :'run_id'
ORDER BY id DESC
LIMIT 1;
SQL
if jq -e '
.state == "completed" and
.attempt == 2 and
.max_attempts == 2 and
.error_count == 1 and
.worker == "WhoNeedHelp.Workers.LocalRetryProbe"
' "$output_dir/retry-job.json" >/dev/null 2>&1; then
break
fi
sleep 1
done
if ! jq -e '
.state == "completed" and
.attempt == 2 and
.max_attempts == 2 and
.error_count == 1 and
.worker == "WhoNeedHelp.Workers.LocalRetryProbe"
' "$output_dir/retry-job.json" >/dev/null; then
echo "The isolated Oban retry probe did not complete its second attempt." >&2
exit 1
fi
delete_retry_job "$output_dir/retry-job-cleanup.txt"
if [[ "$(tail -n 1 "$output_dir/retry-job-cleanup.txt")" != "1" ]]; then
echo "The exact resilience probe job was not removed." >&2
exit 1
fi
retry_job_inserted=false
stop_probe
probe_pid=
jq -s '{
samples: length,
failures: (map(select(.status != "200")) | length),
nodes: (map(.body | fromjson? | .node) | map(select(. != null)) | unique)
}' "$probe_log" >"$output_dir/readiness-summary.json"
if ! jq -e '.samples > 0 and .failures == 0 and (.nodes | length) >= 2' \
"$output_dir/readiness-summary.json" >/dev/null; then
echo "Readiness was unavailable or did not reach multiple web replicas." >&2
exit 1
fi
"${compose[@]}" ps -a >"$output_dir/compose-after.txt"
"${compose[@]}" logs --since "$run_started_at" proxy web worker \
>"$output_dir/application.log" 2>&1
trap - EXIT HUP INT TERM
printf 'Resilience evidence: %s\n' "$output_dir"

View File

@ -0,0 +1,24 @@
defmodule WhoNeedHelp.Workers.LocalRetryProbeTest do
use ExUnit.Case, async: true
use Oban.Testing, repo: WhoNeedHelp.Repo
alias WhoNeedHelp.Workers.LocalRetryProbe
@args %{
"confirmation" => "isolated-local-resilience-probe",
"run_id" => "unit-probe"
}
test "fails the first attempt and succeeds on retry" do
assert {:error, :expected_first_attempt_failure} =
perform_job(LocalRetryProbe, @args, attempt: 1)
assert :ok = perform_job(LocalRetryProbe, @args, attempt: 2)
assert 1 = LocalRetryProbe.backoff(%Oban.Job{attempt: 1})
end
test "cancels jobs without the local confirmation" do
assert {:cancel, :invalid_local_resilience_probe} =
perform_job(LocalRetryProbe, %{"run_id" => "unit-probe"}, attempt: 1)
end
end