test: verify replica failure and rolling recovery
This commit is contained in:
parent
909e16c969
commit
594065deb8
|
|
@ -4,6 +4,7 @@
|
||||||
HTTP_PORT=0
|
HTTP_PORT=0
|
||||||
MAILPIT_PORT=0
|
MAILPIT_PORT=0
|
||||||
TRAEFIK_TRUSTED_IPS=127.0.0.1/32
|
TRAEFIK_TRUSTED_IPS=127.0.0.1/32
|
||||||
|
TRAEFIK_RETRY_ATTEMPTS=3
|
||||||
TRAEFIK_PROJECT_CONSTRAINT=GENERATED_UNIQUE_E2E_PROJECT
|
TRAEFIK_PROJECT_CONSTRAINT=GENERATED_UNIQUE_E2E_PROJECT
|
||||||
TRAEFIK_APP_NAME=GENERATED_UNIQUE_E2E_ROUTER
|
TRAEFIK_APP_NAME=GENERATED_UNIQUE_E2E_ROUTER
|
||||||
TRAEFIK_DOCKER_NETWORK=GENERATED_UNIQUE_E2E_NETWORK
|
TRAEFIK_DOCKER_NETWORK=GENERATED_UNIQUE_E2E_NETWORK
|
||||||
|
|
|
||||||
|
|
@ -5,6 +5,7 @@ MAILPIT_PORT=8027
|
||||||
# Comma-separated proxy IP/CIDR values whose X-Forwarded-* headers Traefik
|
# Comma-separated proxy IP/CIDR values whose X-Forwarded-* headers Traefik
|
||||||
# accepts. Keep loopback locally; set the exact VPN proxy address for staging.
|
# accepts. Keep loopback locally; set the exact VPN proxy address for staging.
|
||||||
TRAEFIK_TRUSTED_IPS=127.0.0.1/32
|
TRAEFIK_TRUSTED_IPS=127.0.0.1/32
|
||||||
|
TRAEFIK_RETRY_ATTEMPTS=3
|
||||||
# Docker-provider isolation and names. A second Compose project must use its
|
# Docker-provider isolation and names. A second Compose project must use its
|
||||||
# own project constraint, router/service name, Docker network, and Host rule.
|
# own project constraint, router/service name, Docker network, and Host rule.
|
||||||
TRAEFIK_PROJECT_CONSTRAINT=who_need_help
|
TRAEFIK_PROJECT_CONSTRAINT=who_need_help
|
||||||
|
|
|
||||||
|
|
@ -10,6 +10,7 @@ PHX_HOST=load.local
|
||||||
PHX_SCHEME=https
|
PHX_SCHEME=https
|
||||||
PHX_URL_PORT=443
|
PHX_URL_PORT=443
|
||||||
TRAEFIK_TRUSTED_IPS=127.0.0.1/32
|
TRAEFIK_TRUSTED_IPS=127.0.0.1/32
|
||||||
|
TRAEFIK_RETRY_ATTEMPTS=3
|
||||||
TRAEFIK_PROJECT_CONSTRAINT=who_need_help_load
|
TRAEFIK_PROJECT_CONSTRAINT=who_need_help_load
|
||||||
TRAEFIK_APP_NAME=who-need-help-load
|
TRAEFIK_APP_NAME=who-need-help-load
|
||||||
TRAEFIK_DOCKER_NETWORK=who_need_help_load_internal
|
TRAEFIK_DOCKER_NETWORK=who_need_help_load_internal
|
||||||
|
|
@ -55,3 +56,6 @@ LOAD_AUTH_VUS=8
|
||||||
LOAD_AUTH_WS_TIMEOUT_MS=5000
|
LOAD_AUTH_WS_TIMEOUT_MS=5000
|
||||||
LOAD_AUTH_THINK_SECONDS=0.1
|
LOAD_AUTH_THINK_SECONDS=0.1
|
||||||
LOAD_FIXTURE_PASSWORD=GENERATE_LOAD_FIXTURE_PASSWORD
|
LOAD_FIXTURE_PASSWORD=GENERATE_LOAD_FIXTURE_PASSWORD
|
||||||
|
LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS=120
|
||||||
|
LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS=0.05
|
||||||
|
LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS=2
|
||||||
|
|
|
||||||
17
README.md
17
README.md
|
|
@ -270,6 +270,23 @@ that dump. After a successful rollout it also removes the obsolete chart Secret
|
||||||
and only the local Helm history revisions that stored the former inline
|
and only the local Helm history revisions that stored the former inline
|
||||||
credential fields.
|
credential fields.
|
||||||
|
|
||||||
|
Exercise the verified local rolling-update path without recreating PostGIS or
|
||||||
|
the Secret:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./scripts/kind-rolling-verify.sh local-kind-rollout
|
||||||
|
```
|
||||||
|
|
||||||
|
Compose crash/replacement and Oban retry checks use the separate load project:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./scripts/load-stack-up.sh
|
||||||
|
./scripts/load-resilience-run.sh local-resilience
|
||||||
|
```
|
||||||
|
|
||||||
|
Both scripts retain ignored evidence under `output/resilience/`; their exact
|
||||||
|
mutation and cleanup boundaries are documented in the operations runbook.
|
||||||
|
|
||||||
For an external cluster, provide a real PostgreSQL/PostGIS service and a
|
For an external cluster, provide a real PostgreSQL/PostGIS service and a
|
||||||
pre-created Secret through required `existingSecret`; the chart never renders
|
pre-created Secret through required `existingSecret`; the chart never renders
|
||||||
credentials from tracked values. The Secret must contain `DATABASE_URL`,
|
credentials from tracked values. The Secret must contain `DATABASE_URL`,
|
||||||
|
|
|
||||||
|
|
@ -12,12 +12,12 @@ services:
|
||||||
web:
|
web:
|
||||||
image: who-need-help:load
|
image: who-need-help:load
|
||||||
labels:
|
labels:
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME}.middlewares=${TRAEFIK_APP_NAME}-forwarded
|
- traefik.http.routers.${TRAEFIK_APP_NAME}.middlewares=${TRAEFIK_APP_NAME}-forwarded,${TRAEFIK_APP_NAME}-retry
|
||||||
- traefik.http.middlewares.${TRAEFIK_APP_NAME}-forwarded.headers.customrequestheaders.X-Forwarded-Proto=https
|
- traefik.http.middlewares.${TRAEFIK_APP_NAME}-forwarded.headers.customrequestheaders.X-Forwarded-Proto=https
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.rule=${TRAEFIK_ROUTER_RULE}
|
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.rule=${TRAEFIK_ROUTER_RULE}
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.entrypoints=websecure
|
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.entrypoints=websecure
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.service=${TRAEFIK_APP_NAME}
|
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.service=${TRAEFIK_APP_NAME}
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.middlewares=${TRAEFIK_APP_NAME}-forwarded
|
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.middlewares=${TRAEFIK_APP_NAME}-forwarded,${TRAEFIK_APP_NAME}-retry
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.tls=true
|
- traefik.http.routers.${TRAEFIK_APP_NAME}-tls.tls=true
|
||||||
|
|
||||||
worker:
|
worker:
|
||||||
|
|
|
||||||
|
|
@ -101,6 +101,8 @@ services:
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.rule=${TRAEFIK_ROUTER_RULE:-PathPrefix(`/`)}
|
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.rule=${TRAEFIK_ROUTER_RULE:-PathPrefix(`/`)}
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.entrypoints=web
|
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.entrypoints=web
|
||||||
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.service=${TRAEFIK_APP_NAME:-who-need-help}
|
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.service=${TRAEFIK_APP_NAME:-who-need-help}
|
||||||
|
- traefik.http.routers.${TRAEFIK_APP_NAME:-who-need-help}.middlewares=${TRAEFIK_APP_NAME:-who-need-help}-retry
|
||||||
|
- traefik.http.middlewares.${TRAEFIK_APP_NAME:-who-need-help}-retry.retry.attempts=${TRAEFIK_RETRY_ATTEMPTS:-3}
|
||||||
- traefik.http.services.${TRAEFIK_APP_NAME:-who-need-help}.loadbalancer.server.port=4000
|
- traefik.http.services.${TRAEFIK_APP_NAME:-who-need-help}.loadbalancer.server.port=4000
|
||||||
healthcheck:
|
healthcheck:
|
||||||
test: ["CMD", "curl", "--fail", "--silent", "http://localhost:4000/healthz/ready"]
|
test: ["CMD", "curl", "--fail", "--silent", "http://localhost:4000/healthz/ready"]
|
||||||
|
|
|
||||||
|
|
@ -10,13 +10,13 @@ item below unless the evidence column explicitly describes a local mock.
|
||||||
| Browser E2E | Manual headed-Chrome scenarios exist; no committed browser suite | A fresh uniquely named Compose project runs two-user urgent help, Activity, moderation, privacy, and error paths; traces are retained on failure; its exact volume is removed |
|
| Browser E2E | Manual headed-Chrome scenarios exist; no committed browser suite | A fresh uniquely named Compose project runs two-user urgent help, Activity, moderation, privacy, and error paths; traces are retained on failure; its exact volume is removed |
|
||||||
| Android UI | Two JVM unit-test files; no `androidTest` source set | Emulator instrumentation covers deep links, permissions, foreground tracking, notification Stop, lifecycle, and network failure |
|
| Android UI | Two JVM unit-test files; no `androidTest` source set | Emulator instrumentation covers deep links, permissions, foreground tracking, notification Stop, lifecycle, and network failure |
|
||||||
| CI and quality | No tracked CI workflow or static/security analysis dependencies | The same containerized gates pass locally and are represented in a validated CI workflow |
|
| CI and quality | No tracked CI workflow or static/security analysis dependencies | The same containerized gates pass locally and are represented in a validated CI workflow |
|
||||||
| Localization and accessibility | Completed locally: product copy and custom validation messages are extracted; EN/UK/RU catalogs and localized category descriptions/structured values are implemented | 508 default and 40 error messages are current; RU/UK have no empty/fuzzy entries; 161 backend tests and all 8 browser specs pass, including locale persistence, keyboard, axe, themes, responsive widths, and reconnect |
|
| Localization and accessibility | Completed locally: product copy and custom validation messages are extracted; EN/UK/RU catalogs and localized category descriptions/structured values are implemented | 508 default and 40 error messages are current; RU/UK have no empty/fuzzy entries; 163 backend tests and all 8 browser specs pass, including locale persistence, keyboard, axe, themes, responsive widths, and reconnect |
|
||||||
| Database scale | Core discovery/chat/moderation lists call unbounded `Repo.all()` | Cursor-bounded queries pass behavior tests and measured `EXPLAIN ANALYZE` checks on an isolated generated dataset |
|
| Database scale | Core discovery/chat/moderation lists call unbounded `Repo.all()` | Cursor-bounded queries pass behavior tests and measured `EXPLAIN ANALYZE` checks on an isolated generated dataset |
|
||||||
| Load and resilience | Public/readiness/heartbeat k6 profile exists | Authenticated writes, chat, tracking, reconnect, rolling replacement, and worker retry profiles pass without touching staging data |
|
| Load and resilience | Public/readiness/heartbeat k6 profile exists | Authenticated writes, chat, tracking, reconnect, rolling replacement, and worker retry profiles pass without touching staging data |
|
||||||
| Observability | Protected Prometheus text endpoint exists | Local Prometheus/Grafana/Alertmanager profile scrapes every replica and an induced isolated failure exercises alert delivery |
|
| Observability | Protected Prometheus text endpoint exists | Local Prometheus/Grafana/Alertmanager profile scrapes every replica and an induced isolated failure exercises alert delivery |
|
||||||
| Backup | Validated local custom-format dump and restore drill exist | An encrypted artifact is uploaded to local S3-compatible MinIO and restored into a fresh database; corruption and interrupted-upload checks fail closed |
|
| Backup | Validated local custom-format dump and restore drill exist | An encrypted artifact is uploaded to local S3-compatible MinIO and restored into a fresh database; corruption and interrupted-upload checks fail closed |
|
||||||
| External boundaries | Mailpit and a fake GitHub strategy cover parts of SMTP/OAuth | Local protocol-level SMTP/OAuth mocks and the applicable push adapter boundary cover success, rejection, retry, replay, and timeout |
|
| External boundaries | Mailpit and a fake GitHub strategy cover parts of SMTP/OAuth | Local protocol-level SMTP/OAuth mocks and the applicable push adapter boundary cover success, rejection, retry, replay, and timeout |
|
||||||
| Final regression | 161 Phoenix tests plus reproducible browser and Android device suites | Browser, Android, API, DB, WebSocket, backup, monitoring, failure, cleanup, docs, and clean Git are verified from the final commits |
|
| Final regression | 163 Phoenix tests plus reproducible browser and Android device suites | Browser, Android, API, DB, WebSocket, backup, monitoring, failure, cleanup, docs, and clean Git are verified from the final commits |
|
||||||
|
|
||||||
The goal remains open while any row lacks reproducible local evidence.
|
The goal remains open while any row lacks reproducible local evidence.
|
||||||
|
|
||||||
|
|
@ -55,7 +55,7 @@ The goal remains open while any row lacks reproducible local evidence.
|
||||||
- The containerized `scripts/quality.sh` gate passes ShellCheck, Hadolint,
|
- The containerized `scripts/quality.sh` gate passes ShellCheck, Hadolint,
|
||||||
actionlint, all Compose renders, Helm lint, a Trivy scan of tracked source and
|
actionlint, all Compose renders, Helm lint, a Trivy scan of tracked source and
|
||||||
the rendered Kubernetes manifest, compiler/xref/Credo/Sobelow/Dialyzer/Hex
|
the rendered Kubernetes manifest, compiler/xref/Credo/Sobelow/Dialyzer/Hex
|
||||||
checks, 161 Phoenix tests, both npm audits, and a Trivy scan of the production
|
checks, 163 Phoenix tests, both npm audits, and a Trivy scan of the production
|
||||||
release image. It creates random one-run database credentials and removes its
|
release image. It creates random one-run database credentials and removes its
|
||||||
exact volume, networks, images, and source snapshot.
|
exact volume, networks, images, and source snapshot.
|
||||||
- The checked-in GitHub Actions workflow runs the same isolated backend/security
|
- The checked-in GitHub Actions workflow runs the same isolated backend/security
|
||||||
|
|
@ -78,5 +78,16 @@ The goal remains open while any row lacks reproducible local evidence.
|
||||||
failure, retained no current positions after stop, passed cross-node PubSub
|
failure, retained no current positions after stop, passed cross-node PubSub
|
||||||
and readiness, and restored every tracked table count after exact fixture
|
and readiness, and restored every tracked table count after exact fixture
|
||||||
cleanup.
|
cleanup.
|
||||||
|
- The same isolated profile now passes deliberate web/worker BEAM crashes,
|
||||||
|
sequential replacement of every replica, exact five-node cluster/PubSub
|
||||||
|
checks, and an Oban job that records one failure before succeeding on its
|
||||||
|
second attempt. A 743-sample readiness probe observed no final HTTP failure
|
||||||
|
and the probe job/domain fixtures were absent afterward.
|
||||||
|
- The project-owned kind cluster also passes a full web/worker rolling restart:
|
||||||
|
all four pod UIDs changed, all replacements became Ready with zero restarts,
|
||||||
|
the four-node cluster/PubSub probe passed, and database counts were unchanged.
|
||||||
|
The local single-node NodePort needed three reconnect attempts across 305
|
||||||
|
ultimately successful samples; this is recorded rather than presented as
|
||||||
|
raw transport continuity.
|
||||||
- The remaining rows above are still pending; this document is not a
|
- The remaining rows above are still pending; this document is not a
|
||||||
completion claim for the entire hardening goal.
|
completion claim for the entire hardening goal.
|
||||||
|
|
|
||||||
|
|
@ -68,6 +68,52 @@ on one connected BEAM node and broadcasts from another.
|
||||||
Health checks do not replace alerting, database backups, restore drills, or
|
Health checks do not replace alerting, database backups, restore drills, or
|
||||||
application-level synthetic checks.
|
application-level synthetic checks.
|
||||||
|
|
||||||
|
## Local failure and rolling-replacement drills
|
||||||
|
|
||||||
|
The isolated load project can exercise process crashes, sequential container
|
||||||
|
replacement, and a real Oban retry without touching the normal Compose project:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./scripts/load-stack-up.sh
|
||||||
|
./scripts/load-resilience-run.sh local-resilience
|
||||||
|
```
|
||||||
|
|
||||||
|
The resilience script refuses `LOAD_PROJECT=who_need_help` and verifies the
|
||||||
|
Compose project/service labels of every container before stopping it. It:
|
||||||
|
|
||||||
|
1. continuously calls readiness through the isolated Traefik route;
|
||||||
|
2. terminates the BEAM process in one web and one worker container and requires
|
||||||
|
Docker's observed restart count to increase;
|
||||||
|
3. removes and replaces each web and worker replica one at a time;
|
||||||
|
4. waits for every configured BEAM node, then runs the cross-node PubSub probe;
|
||||||
|
5. enqueues a side-effect-free local worker that fails its first Oban attempt
|
||||||
|
and succeeds on its second;
|
||||||
|
6. removes that exact Oban row and requires no fixture domain rows to remain.
|
||||||
|
|
||||||
|
Traefik's retry middleware is attached to the HTTP and local TLS routers. Its
|
||||||
|
attempt count is an environment input. Traefik retries transport failures and,
|
||||||
|
with the checked configuration, does not opt in to retrying non-idempotent
|
||||||
|
requests. This reduces a stale-backend window; it is not a claim of production
|
||||||
|
availability.
|
||||||
|
|
||||||
|
For the project-owned kind cluster, run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./scripts/kind-rolling-verify.sh local-kind-rollout
|
||||||
|
```
|
||||||
|
|
||||||
|
That script requires both the kind ownership marker and the control-plane
|
||||||
|
cluster label before invoking `rollout restart`. It changes only the web and
|
||||||
|
worker Deployment pod templates. It snapshots application-table counts before
|
||||||
|
and after, continuously probes the observed Docker mapping for the chart's
|
||||||
|
NodePort, requires all four old pod UIDs to disappear, waits for the exact BEAM
|
||||||
|
peer count, and verifies cross-node PubSub. PostGIS, its hostPath, the
|
||||||
|
Kubernetes Secret, and the namespace are not recreated.
|
||||||
|
|
||||||
|
The rollout timeout, probe interval/timeout/retry count, and cluster-join
|
||||||
|
timeout are experiment inputs. They are not production SLOs or resource
|
||||||
|
requirements.
|
||||||
|
|
||||||
## Protected Prometheus metrics
|
## Protected Prometheus metrics
|
||||||
|
|
||||||
The web role exposes Prometheus text format at `/metrics`. It requires the
|
The web role exposes Prometheus text format at `/metrics`. It requires the
|
||||||
|
|
|
||||||
|
|
@ -173,6 +173,37 @@ Ignored evidence:
|
||||||
|
|
||||||
- `output/performance/auth-final-20260719i/`
|
- `output/performance/auth-final-20260719i/`
|
||||||
|
|
||||||
|
## Observed local resilience drills
|
||||||
|
|
||||||
|
The canonical Compose drill on 2026-07-19 used the isolated 3-web/2-worker
|
||||||
|
profile. One web and one worker BEAM process exited with status 1 and each
|
||||||
|
container's observed restart count increased to 1. All original replicas were
|
||||||
|
then replaced sequentially. The route returned 743 successful readiness
|
||||||
|
responses with zero final failure and responses from all three web nodes.
|
||||||
|
After replacement, the observed cluster contained all five BEAM nodes and the
|
||||||
|
cross-node PubSub probe passed.
|
||||||
|
|
||||||
|
The local Oban probe completed with state `completed`, attempt `2`,
|
||||||
|
`max_attempts=2`, and exactly one recorded first-attempt error. Its exact row
|
||||||
|
was removed afterward. The load database then contained zero probe jobs, users,
|
||||||
|
requests, messages, and tracking sessions. Run-scoped available logs contained
|
||||||
|
no unexpected application error, warning, HTTP 4xx/5xx, or database deadlock.
|
||||||
|
|
||||||
|
The canonical kind drill rolled both 2-replica Deployments from revision 17 to
|
||||||
|
18 with the chart's observed `maxUnavailable=0` and `maxSurge=1`. All four pod
|
||||||
|
UIDs changed, all replacements were Ready with zero container restart, the
|
||||||
|
four-node BEAM cluster and PubSub probe passed, and the application-table count
|
||||||
|
diff was empty. Of 305 readiness samples, all ultimately returned HTTP 200.
|
||||||
|
Two samples needed three transport retries in total while kind's single-node
|
||||||
|
NodePort reset connections during endpoint changes. Those retries are retained
|
||||||
|
in evidence rather than reported as uninterrupted raw TCP connections. This is
|
||||||
|
a local kind observation, not a production availability guarantee.
|
||||||
|
|
||||||
|
Ignored evidence:
|
||||||
|
|
||||||
|
- `output/resilience/compose-resilience-canonical-20260719f/`
|
||||||
|
- `output/resilience/kind-rollout-canonical-20260719c/`
|
||||||
|
|
||||||
Stop the isolated containers without deleting their database volume:
|
Stop the isolated containers without deleting their database volume:
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
|
|
|
||||||
|
|
@ -16,14 +16,14 @@ results from product limits and unknown production properties.
|
||||||
| Consent-driven live tracking | Implemented and cross-client verified | On API 37, Android started `TrackingService` as a location foreground service with a persistent Stop notification. After Home minimized the Activity, an emulator coordinate change reached PostGIS. Notification Stop removed the service, notification, active session, and raw position. | Browsers stop with the page. Android has no `ACCESS_BACKGROUND_LOCATION`, unattended start, or route history. |
|
| Consent-driven live tracking | Implemented and cross-client verified | On API 37, Android started `TrackingService` as a location foreground service with a persistent Stop notification. After Home minimized the Activity, an emulator coordinate change reached PostGIS. Notification Stop removed the service, notification, active session, and raw position. | Browsers stop with the page. Android has no `ACCESS_BACKGROUND_LOCATION`, unattended start, or route history. |
|
||||||
| Privacy settings | Implemented and browser-verified | The profile exposed hidden, approximate public, exact for active match, and explicit exact-public options. Blocking and current-position cleanup have automated tests. | Exact public location remains a user opt-in; legal privacy and retention text still requires jurisdiction-specific review before launch. |
|
| Privacy settings | Implemented and browser-verified | The profile exposed hidden, approximate public, exact for active match, and explicit exact-public options. Blocking and current-position cleanup have automated tests. | Exact public location remains a user opt-in; legal privacy and retention text still requires jurisdiction-specific review before launch. |
|
||||||
| Reputation and anti-abuse | Implemented at MVP level | Handover codes, two-party completion, double-blind reviews, unique-counterpart ranking, optional movement/proximity evidence, reports, blocks, abuse signals, and moderator audit paths have automated tests. | The system is not bot-proof and does not claim identity verification. No punitive numeric policy is enabled without measured and approved thresholds. |
|
| Reputation and anti-abuse | Implemented at MVP level | Handover codes, two-party completion, double-blind reviews, unique-counterpart ranking, optional movement/proximity evidence, reports, blocks, abuse signals, and moderator audit paths have automated tests. | The system is not bot-proof and does not claim identity verification. No punitive numeric policy is enabled without measured and approved thresholds. |
|
||||||
| Social profiles | Manual links implemented; GitHub verification implemented and automated-tested | Manual links cannot set verification fields. The optional GitHub flow uses state, PKCE, a user-bound one-time session, unique provider ownership, and an audit record; 161 tests pass, including callback replay/state checks. No access-token field exists and the controller receives only normalized identity attributes. | The staging operator has not supplied GitHub OAuth credentials, so the real external provider redirect/callback remains disabled and has not been browser-verified. Other providers remain manual/unverified. |
|
| Social profiles | Manual links implemented; GitHub verification implemented and automated-tested | Manual links cannot set verification fields. The optional GitHub flow uses state, PKCE, a user-bound one-time session, unique provider ownership, and an audit record; 163 tests pass, including callback replay/state checks. No access-token field exists and the controller receives only normalized identity attributes. | The staging operator has not supplied GitHub OAuth credentials, so the real external provider redirect/callback remains disabled and has not been browser-verified. Other providers remain manual/unverified. |
|
||||||
| Voluntary thanks | Implemented as an external optional link | A helper can expose an optional link after completion; the UI states that the platform does not process the payment. | The platform does not provide payments, escrow, refunds, tax reporting, or payment guarantees. |
|
| Voluntary thanks | Implemented as an external optional link | A helper can expose an optional link after completion; the UI states that the platform does not process the payment. | The platform does not provide payments, escrow, refunds, tax reporting, or payment guarantees. |
|
||||||
| Android client | Local and public-staging clients implemented and emulator-verified | The native packages `org.whoneedhelp.mobile.debug` and `org.whoneedhelp.mobile.staging` launch the same authenticated LiveView app. Public HTTPS login, map, two-way chat, permission prompts, minimized foreground-service location updates, notification Stop, deep-link routing, and server cleanup were exercised on API 37. | Production signing, Play Store publication, verified Android App Links, unattended/background-permission tracking, and iOS are not implemented. |
|
| Android client | Local and public-staging clients implemented and emulator-verified | The native packages `org.whoneedhelp.mobile.debug` and `org.whoneedhelp.mobile.staging` launch the same authenticated LiveView app. Public HTTPS login, map, two-way chat, permission prompts, minimized foreground-service location updates, notification Stop, deep-link routing, and server cleanup were exercised on API 37. | Production signing, Play Store publication, verified Android App Links, unattended/background-permission tracking, and iOS are not implemented. |
|
||||||
| Multiple web/worker instances | Implemented and locally verified | Docker Compose and kind each ran 2 web and 2 worker replicas. The project probes cross-node Phoenix PubSub using different BEAM nodes. Kubernetes web/worker pods were Ready with zero restarts at the final observation. | Local PostGIS is a single instance. Production database HA, backups, and recovery are operator work and are not claimed complete. |
|
| Multiple web/worker instances | Implemented and locally failure/rollout-verified | The isolated Compose profile passed BEAM crashes and sequential replacement with 3 web/2 worker replicas, all five nodes joined, PubSub passed, and 743/743 readiness requests succeeded. The project-owned kind cluster replaced all 2 web/2 worker pod UIDs under `maxUnavailable=0`; all four replacement pods joined and PubSub passed. | Local PostGIS is a single instance. Production database HA, backups, and recovery are operator work and are not claimed complete. |
|
||||||
|
|
||||||
## Reproducible checks
|
## Reproducible checks
|
||||||
|
|
||||||
- The isolated Phoenix suite completed on 2026-07-19 with 161
|
- The isolated Phoenix suite completed on 2026-07-19 with 163
|
||||||
tests and 0 failures after cursor pagination, database aggregation, and the
|
tests and 0 failures after cursor pagination, database aggregation, and the
|
||||||
full localization changes
|
full localization changes
|
||||||
on Elixir 1.20.2 and Erlang/OTP 29.0.3.
|
on Elixir 1.20.2 and Erlang/OTP 29.0.3.
|
||||||
|
|
@ -32,7 +32,7 @@ results from product limits and unknown production properties.
|
||||||
- `./scripts/quality.sh` passed ShellCheck 0.11.0, Hadolint 2.14.0 at warning
|
- `./scripts/quality.sh` passed ShellCheck 0.11.0, Hadolint 2.14.0 at warning
|
||||||
threshold, actionlint 1.7.12, all four Compose renders, Helm lint, Trivy
|
threshold, actionlint 1.7.12, all four Compose renders, Helm lint, Trivy
|
||||||
source/rendered-manifest scanning, xref, Credo high-priority checks, Sobelow
|
source/rendered-manifest scanning, xref, Credo high-priority checks, Sobelow
|
||||||
strict/private checks, Hex audit, 161 Phoenix tests, both npm audits, and the
|
strict/private checks, Hex audit, 163 Phoenix tests, both npm audits, and the
|
||||||
production-image vulnerability scan. The rendered Helm manifest and Debian
|
production-image vulnerability scan. The rendered Helm manifest and Debian
|
||||||
13.6 release image each reported zero HIGH/CRITICAL findings under the
|
13.6 release image each reported zero HIGH/CRITICAL findings under the
|
||||||
configured gates.
|
configured gates.
|
||||||
|
|
@ -67,6 +67,17 @@ results from product limits and unknown production properties.
|
||||||
containers also used read-only root filesystems and dropped capabilities.
|
containers also used read-only root filesystems and dropped capabilities.
|
||||||
The cross-node probe observed all four BEAM nodes, and live/readiness returned
|
The cross-node probe observed all four BEAM nodes, and live/readiness returned
|
||||||
`ok`/`ready`. Database counts remained `0 users / 0 help requests`.
|
`ok`/`ready`. Database counts remained `0 users / 0 help requests`.
|
||||||
|
- The isolated Compose resilience drill observed a restart count increase for
|
||||||
|
one crashed web and worker BEAM process, replaced every replica sequentially,
|
||||||
|
observed all 5 cluster nodes, passed PubSub, and completed a real Oban retry
|
||||||
|
on attempt 2 after one recorded error. Its exact job row and all fixture
|
||||||
|
domain rows were absent afterward; 743 readiness samples had no final
|
||||||
|
failure.
|
||||||
|
- The reproducible kind rolling drill advanced both Deployments from revision
|
||||||
|
17 to 18, replaced all four pod UIDs, observed all 4 BEAM nodes, passed
|
||||||
|
PubSub, and left the database-count diff empty. All 305 readiness samples
|
||||||
|
ultimately returned 200; two samples needed three total reconnect attempts
|
||||||
|
during local single-node NodePort endpoint replacement.
|
||||||
- The committed browser suite passed its 1/1 bootstrap and all 8/8 Chromium
|
- The committed browser suite passed its 1/1 bootstrap and all 8/8 Chromium
|
||||||
specs against a fresh PostGIS volume with two web and two worker replicas on
|
specs against a fresh PostGIS volume with two web and two worker replicas on
|
||||||
2026-07-19. The retained successful-run artifact directory is
|
2026-07-19. The retained successful-run artifact directory is
|
||||||
|
|
|
||||||
36
lib/who_need_help/workers/local_retry_probe.ex
Normal file
36
lib/who_need_help/workers/local_retry_probe.ex
Normal file
|
|
@ -0,0 +1,36 @@
|
||||||
|
defmodule WhoNeedHelp.Workers.LocalRetryProbe do
|
||||||
|
@moduledoc """
|
||||||
|
A side-effect-free worker used by the isolated local resilience profile.
|
||||||
|
|
||||||
|
The first execution returns an expected error and the second succeeds. No
|
||||||
|
application flow enqueues this worker; the resilience script requires an
|
||||||
|
explicit confirmation value and removes its exact Oban row after collecting
|
||||||
|
evidence.
|
||||||
|
"""
|
||||||
|
|
||||||
|
use Oban.Worker, queue: :maintenance, max_attempts: 2, tags: ["local-resilience-probe"]
|
||||||
|
|
||||||
|
@confirmation "isolated-local-resilience-probe"
|
||||||
|
|
||||||
|
@impl Oban.Worker
|
||||||
|
def perform(%Oban.Job{
|
||||||
|
args: %{"confirmation" => @confirmation, "run_id" => run_id},
|
||||||
|
attempt: 1
|
||||||
|
})
|
||||||
|
when is_binary(run_id) and run_id != "" do
|
||||||
|
{:error, :expected_first_attempt_failure}
|
||||||
|
end
|
||||||
|
|
||||||
|
def perform(%Oban.Job{
|
||||||
|
args: %{"confirmation" => @confirmation, "run_id" => run_id},
|
||||||
|
attempt: attempt
|
||||||
|
})
|
||||||
|
when is_binary(run_id) and run_id != "" and attempt >= 2 do
|
||||||
|
:ok
|
||||||
|
end
|
||||||
|
|
||||||
|
def perform(_job), do: {:cancel, :invalid_local_resilience_probe}
|
||||||
|
|
||||||
|
@impl Oban.Worker
|
||||||
|
def backoff(_job), do: 1
|
||||||
|
end
|
||||||
|
|
@ -7,7 +7,15 @@ target="$ROOT/.env.e2e"
|
||||||
|
|
||||||
if [ -f "$target" ]; then
|
if [ -f "$target" ]; then
|
||||||
chmod 600 "$target"
|
chmod 600 "$target"
|
||||||
echo ".env.e2e already exists; no secret was changed."
|
|
||||||
|
if ! grep -q '^TRAEFIK_RETRY_ATTEMPTS=' "$target"; then
|
||||||
|
printf '\nTRAEFIK_RETRY_ATTEMPTS=3\n' >>"$target"
|
||||||
|
chmod 600 "$target"
|
||||||
|
echo "Added the missing Traefik retry input; no E2E secret was changed."
|
||||||
|
else
|
||||||
|
echo ".env.e2e already exists; no secret or experiment input was changed."
|
||||||
|
fi
|
||||||
|
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|
@ -21,6 +29,7 @@ cat >"$target" <<EOF
|
||||||
HTTP_PORT=0
|
HTTP_PORT=0
|
||||||
MAILPIT_PORT=0
|
MAILPIT_PORT=0
|
||||||
TRAEFIK_TRUSTED_IPS=127.0.0.1/32
|
TRAEFIK_TRUSTED_IPS=127.0.0.1/32
|
||||||
|
TRAEFIK_RETRY_ATTEMPTS=3
|
||||||
TRAEFIK_PROJECT_CONSTRAINT=generated-per-run
|
TRAEFIK_PROJECT_CONSTRAINT=generated-per-run
|
||||||
TRAEFIK_APP_NAME=generated-per-run
|
TRAEFIK_APP_NAME=generated-per-run
|
||||||
TRAEFIK_DOCKER_NETWORK=generated-per-run
|
TRAEFIK_DOCKER_NETWORK=generated-per-run
|
||||||
|
|
|
||||||
|
|
@ -15,23 +15,72 @@ done
|
||||||
if [ -f "$ENV_FILE" ]; then
|
if [ -f "$ENV_FILE" ]; then
|
||||||
chmod 600 "$ENV_FILE"
|
chmod 600 "$ENV_FILE"
|
||||||
|
|
||||||
if grep -q '^LOAD_FIXTURE_PASSWORD=' "$ENV_FILE"; then
|
needs_fixture_password=true
|
||||||
|
needs_resilience_timeout=true
|
||||||
|
needs_resilience_interval=true
|
||||||
|
needs_resilience_request_timeout=true
|
||||||
|
needs_traefik_retry_attempts=true
|
||||||
|
|
||||||
|
grep -q '^LOAD_FIXTURE_PASSWORD=' "$ENV_FILE" && needs_fixture_password=false
|
||||||
|
grep -q '^LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS=' "$ENV_FILE" &&
|
||||||
|
needs_resilience_timeout=false
|
||||||
|
grep -q '^LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS=' "$ENV_FILE" &&
|
||||||
|
needs_resilience_interval=false
|
||||||
|
grep -q '^LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS=' "$ENV_FILE" &&
|
||||||
|
needs_resilience_request_timeout=false
|
||||||
|
grep -q '^TRAEFIK_RETRY_ATTEMPTS=' "$ENV_FILE" && needs_traefik_retry_attempts=false
|
||||||
|
|
||||||
|
if [ "$needs_fixture_password" = false ] &&
|
||||||
|
[ "$needs_resilience_timeout" = false ] &&
|
||||||
|
[ "$needs_resilience_interval" = false ] &&
|
||||||
|
[ "$needs_resilience_request_timeout" = false ] &&
|
||||||
|
[ "$needs_traefik_retry_attempts" = false ]; then
|
||||||
echo ".env.load already exists; no secret or experiment input was changed."
|
echo ".env.load already exists; no secret or experiment input was changed."
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
umask 077
|
umask 077
|
||||||
load_fixture_password=$(openssl rand -hex 24)
|
load_fixture_password=
|
||||||
|
|
||||||
|
if [ "$needs_fixture_password" = true ]; then
|
||||||
|
load_fixture_password=$(openssl rand -hex 24)
|
||||||
|
fi
|
||||||
|
|
||||||
{
|
{
|
||||||
printf '\n# Added by the authenticated-load profile upgrade.\n'
|
if [ "$needs_fixture_password" = true ]; then
|
||||||
printf 'LOAD_AUTH_VUS=8\n'
|
printf '\n# Added by the authenticated-load profile upgrade.\n'
|
||||||
printf 'LOAD_AUTH_WS_TIMEOUT_MS=5000\n'
|
printf 'LOAD_AUTH_VUS=8\n'
|
||||||
printf 'LOAD_AUTH_THINK_SECONDS=0.1\n'
|
printf 'LOAD_AUTH_WS_TIMEOUT_MS=5000\n'
|
||||||
printf 'LOAD_FIXTURE_PASSWORD=%s\n' "$load_fixture_password"
|
printf 'LOAD_AUTH_THINK_SECONDS=0.1\n'
|
||||||
|
printf 'LOAD_FIXTURE_PASSWORD=%s\n' "$load_fixture_password"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ "$needs_resilience_timeout" = true ] ||
|
||||||
|
[ "$needs_resilience_interval" = true ]; then
|
||||||
|
printf '\n# Added by the local resilience-profile upgrade.\n'
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ "$needs_resilience_timeout" = true ]; then
|
||||||
|
printf 'LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS=120\n'
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ "$needs_resilience_interval" = true ]; then
|
||||||
|
printf 'LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS=0.05\n'
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ "$needs_resilience_request_timeout" = true ]; then
|
||||||
|
printf 'LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS=2\n'
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ "$needs_traefik_retry_attempts" = true ]; then
|
||||||
|
printf 'TRAEFIK_RETRY_ATTEMPTS=3\n'
|
||||||
|
fi
|
||||||
} >>"$ENV_FILE"
|
} >>"$ENV_FILE"
|
||||||
chmod 600 "$ENV_FILE"
|
chmod 600 "$ENV_FILE"
|
||||||
unset load_fixture_password
|
unset load_fixture_password needs_fixture_password needs_resilience_timeout \
|
||||||
echo "Added authenticated-load inputs and a random fixture password to ignored .env.load."
|
needs_resilience_interval needs_resilience_request_timeout \
|
||||||
|
needs_traefik_retry_attempts
|
||||||
|
echo "Added missing authenticated-load/resilience inputs to ignored .env.load."
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|
|
||||||
417
scripts/kind-rolling-verify.sh
Executable file
417
scripts/kind-rolling-verify.sh
Executable file
|
|
@ -0,0 +1,417 @@
|
||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)
|
||||||
|
KUBECTL="$ROOT/.tools/bin/kubectl"
|
||||||
|
CONTEXT=kind-who-need-help
|
||||||
|
CLUSTER_CONTAINER=who-need-help-control-plane
|
||||||
|
NAMESPACE=who-need-help
|
||||||
|
WEB_DEPLOYMENT=who-need-help-who-need-help-web
|
||||||
|
WORKER_DEPLOYMENT=who-need-help-who-need-help-worker
|
||||||
|
SERVICE=who-need-help-who-need-help
|
||||||
|
OWNERSHIP_MARKER="$ROOT/.tools/who-need-help.owned"
|
||||||
|
ROLLOUT_TIMEOUT=${KIND_ROLLOUT_TIMEOUT:-180s}
|
||||||
|
PROBE_INTERVAL=${KIND_ROLLOUT_PROBE_INTERVAL_SECONDS:-0.05}
|
||||||
|
PROBE_TIMEOUT=${KIND_ROLLOUT_PROBE_TIMEOUT_SECONDS:-2}
|
||||||
|
PROBE_RETRIES=${KIND_ROLLOUT_PROBE_RETRIES:-2}
|
||||||
|
CLUSTER_JOIN_TIMEOUT=${KIND_CLUSTER_JOIN_TIMEOUT_SECONDS:-120}
|
||||||
|
LABEL=${1:-"kind-rollout-$(date -u +%Y%m%dT%H%M%SZ)"}
|
||||||
|
|
||||||
|
if [[ ! -x "$KUBECTL" ]]; then
|
||||||
|
echo "The project-owned kubectl is missing. Run scripts/bootstrap-kubernetes-tools.sh." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! -f "$OWNERSHIP_MARKER" ]]; then
|
||||||
|
echo "The kind ownership marker is missing; refusing to mutate the cluster." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! "$LABEL" =~ ^[A-Za-z0-9._-]+$ ]]; then
|
||||||
|
echo "Run label may contain only letters, numbers, dot, underscore, and dash." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! "$ROLLOUT_TIMEOUT" =~ ^[1-9][0-9]*[smh]$ ]]; then
|
||||||
|
echo "KIND_ROLLOUT_TIMEOUT must be a positive Kubernetes duration in s, m, or h." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if ! awk -v value="$PROBE_INTERVAL" \
|
||||||
|
'BEGIN {exit !(value ~ /^[0-9]+([.][0-9]+)?$/ && value > 0)}'; then
|
||||||
|
echo "KIND_ROLLOUT_PROBE_INTERVAL_SECONDS must be greater than zero." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! "$PROBE_TIMEOUT" =~ ^[1-9][0-9]*$ ]]; then
|
||||||
|
echo "KIND_ROLLOUT_PROBE_TIMEOUT_SECONDS must be a positive integer." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! "$PROBE_RETRIES" =~ ^[0-9]+$ ]]; then
|
||||||
|
echo "KIND_ROLLOUT_PROBE_RETRIES must be a non-negative integer." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! "$CLUSTER_JOIN_TIMEOUT" =~ ^[1-9][0-9]*$ ]]; then
|
||||||
|
echo "KIND_CLUSTER_JOIN_TIMEOUT_SECONDS must be a positive integer." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
kube=("$KUBECTL" --context "$CONTEXT" --namespace "$NAMESPACE")
|
||||||
|
|
||||||
|
if ! "$KUBECTL" config get-contexts -o name | grep -Fxq "$CONTEXT"; then
|
||||||
|
echo "The expected local kind context does not exist." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "$(docker inspect --format '{{index .Config.Labels "io.x-k8s.kind.cluster"}}' \
|
||||||
|
"$CLUSTER_CONTAINER")" != "who-need-help" ]]; then
|
||||||
|
echo "The kind control-plane container does not belong to this project cluster." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
output_dir="$ROOT/output/resilience/$LABEL"
|
||||||
|
mkdir -p "$output_dir"
|
||||||
|
chmod 700 "$ROOT/output" "$ROOT/output/resilience" "$output_dir"
|
||||||
|
probe_marker="$output_dir/.probe-running"
|
||||||
|
probe_log="$output_dir/readiness.jsonl"
|
||||||
|
run_started_at=$(date -u +%Y-%m-%dT%H:%M:%SZ)
|
||||||
|
touch "$probe_marker"
|
||||||
|
|
||||||
|
deployment_snapshot() {
|
||||||
|
"${kube[@]}" get deployment "$WEB_DEPLOYMENT" "$WORKER_DEPLOYMENT" -o json |
|
||||||
|
jq '[
|
||||||
|
.items[] | {
|
||||||
|
name: .metadata.name,
|
||||||
|
generation: .metadata.generation,
|
||||||
|
revision: .metadata.annotations["deployment.kubernetes.io/revision"],
|
||||||
|
replicas: .spec.replicas,
|
||||||
|
ready: .status.readyReplicas,
|
||||||
|
available: .status.availableReplicas,
|
||||||
|
updated: .status.updatedReplicas,
|
||||||
|
strategy: .spec.strategy,
|
||||||
|
image: .spec.template.spec.containers[0].image
|
||||||
|
}
|
||||||
|
]' >"$1"
|
||||||
|
}
|
||||||
|
|
||||||
|
database_snapshot() {
|
||||||
|
# Variables are intentionally expanded inside the PostGIS container.
|
||||||
|
# shellcheck disable=SC2016
|
||||||
|
"${kube[@]}" exec -i postgis-0 -- sh -c \
|
||||||
|
'psql --no-psqlrc --tuples-only --no-align --set ON_ERROR_STOP=1 \
|
||||||
|
--username "$POSTGRES_USER" --dbname "$POSTGRES_DB"' >"$1" <<'SQL'
|
||||||
|
BEGIN READ ONLY;
|
||||||
|
SELECT 'users' AS table_name, count(*) AS row_count FROM users
|
||||||
|
UNION ALL SELECT 'users_tokens', count(*) FROM users_tokens
|
||||||
|
UNION ALL SELECT 'help_requests', count(*) FROM help_requests
|
||||||
|
UNION ALL SELECT 'messages', count(*) FROM messages
|
||||||
|
UNION ALL SELECT 'categories', count(*) FROM categories
|
||||||
|
UNION ALL SELECT 'help_assignments', count(*) FROM help_assignments
|
||||||
|
UNION ALL SELECT 'activities', count(*) FROM activities
|
||||||
|
UNION ALL SELECT 'reports', count(*) FROM reports
|
||||||
|
UNION ALL SELECT 'social_identities', count(*) FROM social_identities
|
||||||
|
UNION ALL SELECT 'tracking_sessions', count(*) FROM tracking_sessions
|
||||||
|
UNION ALL SELECT 'tracking_positions', count(*) FROM tracking_positions
|
||||||
|
UNION ALL SELECT 'schema_migrations', count(*) FROM schema_migrations
|
||||||
|
ORDER BY table_name;
|
||||||
|
COMMIT;
|
||||||
|
SQL
|
||||||
|
}
|
||||||
|
|
||||||
|
pod_snapshot() {
|
||||||
|
"${kube[@]}" get pods \
|
||||||
|
-l app.kubernetes.io/instance=who-need-help \
|
||||||
|
-o json |
|
||||||
|
jq '[
|
||||||
|
.items[]
|
||||||
|
| select(
|
||||||
|
.metadata.labels["app.kubernetes.io/component"] == "web" or
|
||||||
|
.metadata.labels["app.kubernetes.io/component"] == "worker"
|
||||||
|
)
|
||||||
|
| {
|
||||||
|
name: .metadata.name,
|
||||||
|
uid: .metadata.uid,
|
||||||
|
component: .metadata.labels["app.kubernetes.io/component"],
|
||||||
|
ready: ([.status.containerStatuses[]?.ready] | all),
|
||||||
|
restarts: ([.status.containerStatuses[]?.restartCount] | add // 0)
|
||||||
|
}
|
||||||
|
] | sort_by(.component, .name)' >"$1"
|
||||||
|
}
|
||||||
|
|
||||||
|
stop_probe() {
|
||||||
|
unlink "$probe_marker" 2>/dev/null || true
|
||||||
|
|
||||||
|
if [[ -n "${probe_pid:-}" ]]; then
|
||||||
|
wait "$probe_pid" 2>/dev/null || true
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
local status=$?
|
||||||
|
trap - EXIT HUP INT TERM
|
||||||
|
stop_probe
|
||||||
|
exit "$status"
|
||||||
|
}
|
||||||
|
|
||||||
|
trap cleanup EXIT HUP INT TERM
|
||||||
|
|
||||||
|
deployment_snapshot "$output_dir/deployments-before.json"
|
||||||
|
|
||||||
|
if ! jq -e '
|
||||||
|
length == 2 and
|
||||||
|
all(
|
||||||
|
.replicas >= 2 and
|
||||||
|
.ready == .replicas and
|
||||||
|
.available == .replicas and
|
||||||
|
.updated == .replicas and
|
||||||
|
.strategy.type == "RollingUpdate" and
|
||||||
|
.strategy.rollingUpdate.maxUnavailable == 0 and
|
||||||
|
.strategy.rollingUpdate.maxSurge == 1
|
||||||
|
)
|
||||||
|
' "$output_dir/deployments-before.json" >/dev/null; then
|
||||||
|
echo "The local deployments are not ready for the recorded rolling strategy." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
database_snapshot "$output_dir/database-before.txt"
|
||||||
|
pod_snapshot "$output_dir/pods-before.json"
|
||||||
|
|
||||||
|
node_port=$(
|
||||||
|
"${kube[@]}" get service "$SERVICE" \
|
||||||
|
-o jsonpath='{.spec.ports[?(@.name=="http")].nodePort}'
|
||||||
|
)
|
||||||
|
published=$(
|
||||||
|
docker port "$CLUSTER_CONTAINER" "${node_port}/tcp" |
|
||||||
|
head -n 1
|
||||||
|
)
|
||||||
|
host_port=${published##*:}
|
||||||
|
|
||||||
|
if [[ ! "$host_port" =~ ^[1-9][0-9]*$ ]]; then
|
||||||
|
echo "The kind HTTP NodePort has no observed Docker host mapping." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
base_url="http://127.0.0.1:$host_port"
|
||||||
|
|
||||||
|
sample_readiness() {
|
||||||
|
while [[ -e "$probe_marker" ]]; do
|
||||||
|
observed_at=$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ)
|
||||||
|
body_file="$output_dir/kind-readiness-body.$$"
|
||||||
|
error_file="$output_dir/kind-readiness-error.$$"
|
||||||
|
set +e
|
||||||
|
result=$(
|
||||||
|
curl --silent --show-error \
|
||||||
|
--max-time "$PROBE_TIMEOUT" \
|
||||||
|
--retry "$PROBE_RETRIES" \
|
||||||
|
--retry-all-errors \
|
||||||
|
--retry-connrefused \
|
||||||
|
--retry-delay 0 \
|
||||||
|
--output "$body_file" \
|
||||||
|
--write-out '%{http_code} %{num_retries}' \
|
||||||
|
"$base_url/healthz/ready" 2>"$error_file"
|
||||||
|
)
|
||||||
|
curl_status=$?
|
||||||
|
set -e
|
||||||
|
read -r status retries <<<"$result"
|
||||||
|
|
||||||
|
if [[ "$curl_status" -ne 0 ]]; then
|
||||||
|
status=000
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ -f "$body_file" ]]; then
|
||||||
|
body=$(tr -d '\n' <"$body_file")
|
||||||
|
else
|
||||||
|
body=
|
||||||
|
fi
|
||||||
|
|
||||||
|
error=$(tr -d '\n' <"$error_file")
|
||||||
|
unlink "$body_file" 2>/dev/null || true
|
||||||
|
unlink "$error_file" 2>/dev/null || true
|
||||||
|
jq -cn \
|
||||||
|
--arg observed_at "$observed_at" \
|
||||||
|
--arg status "$status" \
|
||||||
|
--arg body "$body" \
|
||||||
|
--arg error "$error" \
|
||||||
|
--argjson retries "${retries:-0}" \
|
||||||
|
'{
|
||||||
|
observed_at: $observed_at,
|
||||||
|
status: $status,
|
||||||
|
retries: $retries,
|
||||||
|
body: $body,
|
||||||
|
error: $error
|
||||||
|
}' >>"$probe_log"
|
||||||
|
sleep "$PROBE_INTERVAL"
|
||||||
|
done
|
||||||
|
}
|
||||||
|
|
||||||
|
{
|
||||||
|
printf 'observed_at=%s\n' "$run_started_at"
|
||||||
|
printf 'context=%s\n' "$CONTEXT"
|
||||||
|
printf 'namespace=%s\n' "$NAMESPACE"
|
||||||
|
printf 'rollout_timeout=%s\n' "$ROLLOUT_TIMEOUT"
|
||||||
|
printf 'probe_interval_seconds=%s\n' "$PROBE_INTERVAL"
|
||||||
|
printf 'probe_timeout_seconds=%s\n' "$PROBE_TIMEOUT"
|
||||||
|
printf 'probe_retries=%s\n' "$PROBE_RETRIES"
|
||||||
|
printf 'cluster_join_timeout_seconds=%s\n' "$CLUSTER_JOIN_TIMEOUT"
|
||||||
|
printf 'node_port=%s\n' "$node_port"
|
||||||
|
printf 'observed_host_port=%s\n' "$host_port"
|
||||||
|
"$KUBECTL" version --client
|
||||||
|
} >"$output_dir/environment.txt"
|
||||||
|
|
||||||
|
sample_readiness &
|
||||||
|
probe_pid=$!
|
||||||
|
|
||||||
|
"${kube[@]}" rollout restart "deployment/$WEB_DEPLOYMENT" \
|
||||||
|
>"$output_dir/web-rollout-restart.txt"
|
||||||
|
"${kube[@]}" rollout status "deployment/$WEB_DEPLOYMENT" \
|
||||||
|
"--timeout=$ROLLOUT_TIMEOUT" >"$output_dir/web-rollout-status.txt"
|
||||||
|
|
||||||
|
"${kube[@]}" rollout restart "deployment/$WORKER_DEPLOYMENT" \
|
||||||
|
>"$output_dir/worker-rollout-restart.txt"
|
||||||
|
"${kube[@]}" rollout status "deployment/$WORKER_DEPLOYMENT" \
|
||||||
|
"--timeout=$ROLLOUT_TIMEOUT" >"$output_dir/worker-rollout-status.txt"
|
||||||
|
|
||||||
|
"${kube[@]}" wait \
|
||||||
|
--for=condition=Ready \
|
||||||
|
--timeout="$ROLLOUT_TIMEOUT" \
|
||||||
|
pod \
|
||||||
|
-l app.kubernetes.io/instance=who-need-help \
|
||||||
|
>"$output_dir/pods-ready.txt"
|
||||||
|
|
||||||
|
old_pod_deadline=$((SECONDS + CLUSTER_JOIN_TIMEOUT))
|
||||||
|
|
||||||
|
while ((SECONDS < old_pod_deadline)); do
|
||||||
|
"${kube[@]}" get pods \
|
||||||
|
-l app.kubernetes.io/instance=who-need-help \
|
||||||
|
-o json >"$output_dir/pods-current.json"
|
||||||
|
|
||||||
|
retained_count=$(
|
||||||
|
jq \
|
||||||
|
--slurpfile before "$output_dir/pods-before.json" \
|
||||||
|
'[
|
||||||
|
.items[].metadata.uid as $uid
|
||||||
|
| select($before[0] | any(.uid == $uid))
|
||||||
|
] | length' "$output_dir/pods-current.json"
|
||||||
|
)
|
||||||
|
|
||||||
|
if [[ "$retained_count" == "0" ]]; then
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
|
||||||
|
sleep 1
|
||||||
|
done
|
||||||
|
|
||||||
|
unlink "$output_dir/pods-current.json" 2>/dev/null || true
|
||||||
|
|
||||||
|
if [[ "$retained_count" != "0" ]]; then
|
||||||
|
echo "Old application pods did not terminate after the rollout." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
deployment_snapshot "$output_dir/deployments-after.json"
|
||||||
|
pod_snapshot "$output_dir/pods-after.json"
|
||||||
|
database_snapshot "$output_dir/database-after.txt"
|
||||||
|
|
||||||
|
if ! diff -u "$output_dir/database-before.txt" "$output_dir/database-after.txt" \
|
||||||
|
>"$output_dir/database-diff.txt"; then
|
||||||
|
echo "The rolling verification changed tracked database counts." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
jq -n \
|
||||||
|
--slurpfile before "$output_dir/pods-before.json" \
|
||||||
|
--slurpfile after "$output_dir/pods-after.json" \
|
||||||
|
'{
|
||||||
|
before_count: ($before[0] | length),
|
||||||
|
after_count: ($after[0] | length),
|
||||||
|
retained_uids: (
|
||||||
|
[$before[0][].uid] as $old
|
||||||
|
| [$after[0][].uid | select(. as $uid | $old | index($uid))]
|
||||||
|
),
|
||||||
|
after_all_ready: ($after[0] | all(.ready and .restarts == 0))
|
||||||
|
}' >"$output_dir/pod-replacement-summary.json"
|
||||||
|
|
||||||
|
if ! jq -e '
|
||||||
|
.before_count == 4 and
|
||||||
|
.after_count == 4 and
|
||||||
|
.retained_uids == [] and
|
||||||
|
.after_all_ready
|
||||||
|
' "$output_dir/pod-replacement-summary.json" >/dev/null; then
|
||||||
|
echo "The rollout did not replace all four application pods cleanly." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
web_pod=$(
|
||||||
|
"${kube[@]}" get pod \
|
||||||
|
-l app.kubernetes.io/component=web \
|
||||||
|
--field-selector=status.phase=Running \
|
||||||
|
-o jsonpath='{.items[0].metadata.name}'
|
||||||
|
)
|
||||||
|
expected_peers=$(
|
||||||
|
jq '[.[].replicas] | add - 1' "$output_dir/deployments-after.json"
|
||||||
|
)
|
||||||
|
cluster_deadline=$((SECONDS + CLUSTER_JOIN_TIMEOUT))
|
||||||
|
|
||||||
|
while ((SECONDS < cluster_deadline)); do
|
||||||
|
peer_count=$(
|
||||||
|
"${kube[@]}" exec "$web_pod" -- /app/bin/who_need_help rpc \
|
||||||
|
'IO.puts(length(Node.list()))' 2>/dev/null |
|
||||||
|
tail -n 1
|
||||||
|
)
|
||||||
|
|
||||||
|
if [[ "$peer_count" == "$expected_peers" ]]; then
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
|
||||||
|
sleep 1
|
||||||
|
done
|
||||||
|
|
||||||
|
if [[ "$peer_count" != "$expected_peers" ]]; then
|
||||||
|
echo "All rolled application pods did not join the BEAM cluster." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
"${kube[@]}" exec "$web_pod" -- /app/bin/who_need_help rpc \
|
||||||
|
'IO.inspect(%{node: node(), peers: Node.list(), peer_count: length(Node.list())})' \
|
||||||
|
>"$output_dir/cluster-after.txt"
|
||||||
|
"$ROOT/scripts/verify-realtime-cluster.sh" kind \
|
||||||
|
>"$output_dir/pubsub-after.txt"
|
||||||
|
|
||||||
|
stop_probe
|
||||||
|
probe_pid=
|
||||||
|
|
||||||
|
jq -s '{
|
||||||
|
samples: length,
|
||||||
|
failures: (map(select(.status != "200")) | length),
|
||||||
|
retried_samples: (map(select(.retries > 0)) | length),
|
||||||
|
total_retries: (map(.retries) | add),
|
||||||
|
nodes: (map(.body | fromjson? | .node) | map(select(. != null)) | unique)
|
||||||
|
}' "$probe_log" >"$output_dir/readiness-summary.json"
|
||||||
|
|
||||||
|
"${kube[@]}" get pods \
|
||||||
|
-l app.kubernetes.io/component=web \
|
||||||
|
-o json |
|
||||||
|
jq '[.items[].status.podIP | "who_need_help@" + .] | sort' \
|
||||||
|
>"$output_dir/current-web-nodes.json"
|
||||||
|
|
||||||
|
if ! jq -e \
|
||||||
|
--slurpfile current "$output_dir/current-web-nodes.json" '
|
||||||
|
.samples > 0 and
|
||||||
|
.failures == 0 and
|
||||||
|
($current[0] | length) == 2 and
|
||||||
|
(($current[0] - .nodes) | length) == 0
|
||||||
|
' "$output_dir/readiness-summary.json" >/dev/null; then
|
||||||
|
echo "Kind readiness was unavailable or did not reach both web replicas." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
"${kube[@]}" logs \
|
||||||
|
--selector app.kubernetes.io/instance=who-need-help \
|
||||||
|
--all-containers \
|
||||||
|
--since-time "$run_started_at" \
|
||||||
|
--prefix >"$output_dir/application.log" 2>&1
|
||||||
|
|
||||||
|
trap - EXIT HUP INT TERM
|
||||||
|
printf 'Kind rollout evidence: %s\n' "$output_dir"
|
||||||
456
scripts/load-resilience-run.sh
Executable file
456
scripts/load-resilience-run.sh
Executable file
|
|
@ -0,0 +1,456 @@
|
||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)
|
||||||
|
ENV_FILE="$ROOT/.env.load"
|
||||||
|
LABEL=${1:-"resilience-$(date -u +%Y%m%dT%H%M%SZ)"}
|
||||||
|
|
||||||
|
if [[ ! -f "$ENV_FILE" ]]; then
|
||||||
|
echo "Missing $ENV_FILE. Run scripts/ensure-local-load-env.sh first." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
set -a
|
||||||
|
# shellcheck source=/dev/null
|
||||||
|
. "$ENV_FILE"
|
||||||
|
set +a
|
||||||
|
|
||||||
|
for name in LOAD_PROJECT LOAD_HOST LOAD_WEB_REPLICAS LOAD_WORKER_REPLICAS \
|
||||||
|
LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS \
|
||||||
|
LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS \
|
||||||
|
LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS TRAEFIK_RETRY_ATTEMPTS \
|
||||||
|
HTTP_PORT POSTGRES_DB; do
|
||||||
|
if [[ -z "${!name:-}" ]]; then
|
||||||
|
echo "$name is missing from .env.load" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if [[ "$LOAD_PROJECT" == "who_need_help" ]]; then
|
||||||
|
echo "The resilience profile must not use the staging Compose project." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! "$LABEL" =~ ^[A-Za-z0-9._-]+$ ]]; then
|
||||||
|
echo "Run label may contain only letters, numbers, dot, underscore, and dash." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
for name in LOAD_WEB_REPLICAS LOAD_WORKER_REPLICAS \
|
||||||
|
LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS \
|
||||||
|
LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS TRAEFIK_RETRY_ATTEMPTS; do
|
||||||
|
if [[ ! "${!name}" =~ ^[1-9][0-9]*$ ]]; then
|
||||||
|
echo "$name must be a positive integer." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if ! awk -v value="$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS" \
|
||||||
|
'BEGIN {exit !(value ~ /^[0-9]+([.][0-9]+)?$/ && value > 0)}'; then
|
||||||
|
echo "LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS must be greater than zero." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
compose=(
|
||||||
|
docker compose
|
||||||
|
--env-file "$ENV_FILE"
|
||||||
|
-p "$LOAD_PROJECT"
|
||||||
|
-f compose.yaml
|
||||||
|
-f compose.load.yaml
|
||||||
|
)
|
||||||
|
|
||||||
|
output_dir="$ROOT/output/resilience/$LABEL"
|
||||||
|
mkdir -p "$output_dir"
|
||||||
|
chmod 700 "$ROOT/output" "$ROOT/output/resilience" "$output_dir"
|
||||||
|
run_started_at=$(date -u +%Y-%m-%dT%H:%M:%SZ)
|
||||||
|
probe_marker="$output_dir/.probe-running"
|
||||||
|
probe_log="$output_dir/readiness.jsonl"
|
||||||
|
touch "$probe_marker"
|
||||||
|
|
||||||
|
assert_scope() {
|
||||||
|
local container_id=$1
|
||||||
|
local expected_service=$2
|
||||||
|
local observed_project observed_service
|
||||||
|
|
||||||
|
observed_project=$(
|
||||||
|
docker inspect --format '{{index .Config.Labels "com.docker.compose.project"}}' \
|
||||||
|
"$container_id"
|
||||||
|
)
|
||||||
|
observed_service=$(
|
||||||
|
docker inspect --format '{{index .Config.Labels "com.docker.compose.service"}}' \
|
||||||
|
"$container_id"
|
||||||
|
)
|
||||||
|
|
||||||
|
if [[ "$observed_project" != "$LOAD_PROJECT" ||
|
||||||
|
"$observed_service" != "$expected_service" ]]; then
|
||||||
|
echo "Container scope mismatch for $container_id." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
service_ids() {
|
||||||
|
"${compose[@]}" ps -q "$1"
|
||||||
|
}
|
||||||
|
|
||||||
|
wait_for_service() {
|
||||||
|
local service=$1
|
||||||
|
local expected=$2
|
||||||
|
local require_health=$3
|
||||||
|
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
|
||||||
|
|
||||||
|
while ((SECONDS < deadline)); do
|
||||||
|
mapfile -t ids < <(service_ids "$service")
|
||||||
|
|
||||||
|
if [[ "${#ids[@]}" -eq "$expected" ]]; then
|
||||||
|
local ready=true
|
||||||
|
local id state health
|
||||||
|
|
||||||
|
for id in "${ids[@]}"; do
|
||||||
|
state=$(docker inspect --format '{{.State.Status}}' "$id")
|
||||||
|
health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{end}}' "$id")
|
||||||
|
|
||||||
|
if [[ "$state" != "running" ||
|
||||||
|
("$require_health" == "true" && "$health" != "healthy") ]]; then
|
||||||
|
ready=false
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if [[ "$ready" == "true" ]]; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
sleep 1
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "Timed out waiting for $expected ready $service replicas." >&2
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
wait_for_restart() {
|
||||||
|
local container_id=$1
|
||||||
|
local previous_restarts=$2
|
||||||
|
local require_health=$3
|
||||||
|
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
|
||||||
|
|
||||||
|
while ((SECONDS < deadline)); do
|
||||||
|
local state restarts health
|
||||||
|
state=$(docker inspect --format '{{.State.Status}}' "$container_id")
|
||||||
|
restarts=$(docker inspect --format '{{.RestartCount}}' "$container_id")
|
||||||
|
health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{end}}' "$container_id")
|
||||||
|
|
||||||
|
if [[ "$state" == "running" && "$restarts" -gt "$previous_restarts" &&
|
||||||
|
("$require_health" == "false" || "$health" == "healthy") ]]; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
sleep 1
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "Timed out waiting for container $container_id to restart." >&2
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
wait_for_cluster() {
|
||||||
|
local expected_peers=$((LOAD_WEB_REPLICAS + LOAD_WORKER_REPLICAS - 1))
|
||||||
|
local deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
|
||||||
|
|
||||||
|
while ((SECONDS < deadline)); do
|
||||||
|
local target peer_count
|
||||||
|
target=$(service_ids web | head -n 1)
|
||||||
|
assert_scope "$target" web
|
||||||
|
peer_count=$(
|
||||||
|
docker exec "$target" /app/bin/who_need_help rpc \
|
||||||
|
'IO.puts(length(Node.list()))' 2>/dev/null |
|
||||||
|
tail -n 1
|
||||||
|
)
|
||||||
|
|
||||||
|
if [[ "$peer_count" == "$expected_peers" ]]; then
|
||||||
|
docker exec "$target" /app/bin/who_need_help rpc \
|
||||||
|
'IO.inspect(%{node: node(), peers: Node.list(), peer_count: length(Node.list())})'
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
sleep 1
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "Timed out waiting for all local BEAM replicas to join the cluster." >&2
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
sample_readiness() {
|
||||||
|
while [[ -e "$probe_marker" ]]; do
|
||||||
|
observed_at=$(date -u +%Y-%m-%dT%H:%M:%S.%3NZ)
|
||||||
|
body_file="$output_dir/readiness-body.$$"
|
||||||
|
error_file="$output_dir/readiness-error.$$"
|
||||||
|
set +e
|
||||||
|
status=$(
|
||||||
|
curl --silent --show-error \
|
||||||
|
--max-time "$LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS" \
|
||||||
|
--header "Host: $LOAD_HOST" \
|
||||||
|
--output "$body_file" \
|
||||||
|
--write-out '%{http_code}' \
|
||||||
|
"http://localhost:$HTTP_PORT/healthz/ready" 2>"$error_file"
|
||||||
|
)
|
||||||
|
curl_status=$?
|
||||||
|
set -e
|
||||||
|
|
||||||
|
if [[ "$curl_status" -ne 0 ]]; then
|
||||||
|
status=000
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ -f "$body_file" ]]; then
|
||||||
|
body=$(tr -d '\n' <"$body_file")
|
||||||
|
else
|
||||||
|
body=
|
||||||
|
fi
|
||||||
|
|
||||||
|
error=$(tr -d '\n' <"$error_file")
|
||||||
|
unlink "$body_file" 2>/dev/null || true
|
||||||
|
unlink "$error_file" 2>/dev/null || true
|
||||||
|
jq -cn \
|
||||||
|
--arg observed_at "$observed_at" \
|
||||||
|
--arg status "$status" \
|
||||||
|
--arg body "$body" \
|
||||||
|
--arg error "$error" \
|
||||||
|
'{observed_at: $observed_at, status: $status, body: $body, error: $error}' \
|
||||||
|
>>"$probe_log"
|
||||||
|
sleep "$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS"
|
||||||
|
done
|
||||||
|
}
|
||||||
|
|
||||||
|
stop_probe() {
|
||||||
|
unlink "$probe_marker" 2>/dev/null || true
|
||||||
|
|
||||||
|
if [[ -n "${probe_pid:-}" ]]; then
|
||||||
|
wait "$probe_pid" 2>/dev/null || true
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
delete_retry_job() {
|
||||||
|
local destination=$1
|
||||||
|
local target
|
||||||
|
|
||||||
|
target=$(service_ids worker | head -n 1)
|
||||||
|
assert_scope "$target" worker
|
||||||
|
|
||||||
|
docker exec "$target" /app/bin/who_need_help rpc \
|
||||||
|
"import Ecto.Query
|
||||||
|
run_id = \"$LABEL\"
|
||||||
|
{count, _} =
|
||||||
|
WhoNeedHelp.Repo.delete_all(
|
||||||
|
from job in Oban.Job,
|
||||||
|
where:
|
||||||
|
job.worker == \"WhoNeedHelp.Workers.LocalRetryProbe\" and
|
||||||
|
fragment(\"?->>'run_id'\", job.args) == ^run_id
|
||||||
|
)
|
||||||
|
IO.puts(count)" >"$destination"
|
||||||
|
}
|
||||||
|
|
||||||
|
restore_topology() {
|
||||||
|
"${compose[@]}" up -d --no-deps \
|
||||||
|
--scale "web=$LOAD_WEB_REPLICAS" \
|
||||||
|
--scale "worker=$LOAD_WORKER_REPLICAS" \
|
||||||
|
web worker >/dev/null 2>&1 || true
|
||||||
|
}
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
local status=$?
|
||||||
|
local cleanup_status=0
|
||||||
|
|
||||||
|
trap - EXIT HUP INT TERM
|
||||||
|
stop_probe
|
||||||
|
|
||||||
|
set +e
|
||||||
|
restore_topology
|
||||||
|
|
||||||
|
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
|
||||||
|
|
||||||
|
if [[ "${retry_job_inserted:-false}" == "true" ]]; then
|
||||||
|
delete_retry_job "$output_dir/retry-job-cleanup-on-exit.txt"
|
||||||
|
cleanup_status=$?
|
||||||
|
fi
|
||||||
|
set -e
|
||||||
|
|
||||||
|
if [[ "$cleanup_status" -ne 0 ]]; then
|
||||||
|
echo "Resilience probe cleanup failed." >&2
|
||||||
|
status=1
|
||||||
|
fi
|
||||||
|
|
||||||
|
exit "$status"
|
||||||
|
}
|
||||||
|
|
||||||
|
trap cleanup EXIT HUP INT TERM
|
||||||
|
|
||||||
|
wait_for_service web "$LOAD_WEB_REPLICAS" true
|
||||||
|
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
|
||||||
|
|
||||||
|
{
|
||||||
|
printf 'observed_at=%s\n' "$run_started_at"
|
||||||
|
printf 'load_project=%s\n' "$LOAD_PROJECT"
|
||||||
|
printf 'web_replicas=%s\n' "$LOAD_WEB_REPLICAS"
|
||||||
|
printf 'worker_replicas=%s\n' "$LOAD_WORKER_REPLICAS"
|
||||||
|
printf 'recovery_timeout_seconds=%s\n' "$LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS"
|
||||||
|
printf 'probe_interval_seconds=%s\n' "$LOAD_RESILIENCE_PROBE_INTERVAL_SECONDS"
|
||||||
|
printf 'request_timeout_seconds=%s\n' "$LOAD_RESILIENCE_REQUEST_TIMEOUT_SECONDS"
|
||||||
|
printf 'traefik_retry_attempts=%s\n' "$TRAEFIK_RETRY_ATTEMPTS"
|
||||||
|
docker compose version
|
||||||
|
docker info --format 'docker_server={{.ServerVersion}}'
|
||||||
|
} >"$output_dir/environment.txt"
|
||||||
|
|
||||||
|
sample_readiness &
|
||||||
|
probe_pid=$!
|
||||||
|
|
||||||
|
web_target=$(service_ids web | head -n 1)
|
||||||
|
assert_scope "$web_target" web
|
||||||
|
web_restarts=$(docker inspect --format '{{.RestartCount}}' "$web_target")
|
||||||
|
docker exec "$web_target" /app/bin/who_need_help rpc ':erlang.halt(1)' \
|
||||||
|
>"$output_dir/web-kill.txt" 2>&1 || true
|
||||||
|
wait_for_restart "$web_target" "$web_restarts" true
|
||||||
|
docker inspect "$web_target" |
|
||||||
|
jq '.[0] | {
|
||||||
|
id: .Id,
|
||||||
|
project: .Config.Labels["com.docker.compose.project"],
|
||||||
|
service: .Config.Labels["com.docker.compose.service"],
|
||||||
|
restart_count: .RestartCount,
|
||||||
|
state: .State.Status,
|
||||||
|
health: .State.Health.Status
|
||||||
|
}' >"$output_dir/web-restart.json"
|
||||||
|
|
||||||
|
worker_target=$(service_ids worker | head -n 1)
|
||||||
|
assert_scope "$worker_target" worker
|
||||||
|
worker_restarts=$(docker inspect --format '{{.RestartCount}}' "$worker_target")
|
||||||
|
docker exec "$worker_target" /app/bin/who_need_help rpc ':erlang.halt(1)' \
|
||||||
|
>"$output_dir/worker-kill.txt" 2>&1 || true
|
||||||
|
wait_for_restart "$worker_target" "$worker_restarts" false
|
||||||
|
docker inspect "$worker_target" |
|
||||||
|
jq '.[0] | {
|
||||||
|
id: .Id,
|
||||||
|
project: .Config.Labels["com.docker.compose.project"],
|
||||||
|
service: .Config.Labels["com.docker.compose.service"],
|
||||||
|
restart_count: .RestartCount,
|
||||||
|
state: .State.Status
|
||||||
|
}' >"$output_dir/worker-restart.json"
|
||||||
|
|
||||||
|
mapfile -t original_web_ids < <(service_ids web)
|
||||||
|
|
||||||
|
for container_id in "${original_web_ids[@]}"; do
|
||||||
|
assert_scope "$container_id" web
|
||||||
|
{
|
||||||
|
docker stop --time 30 "$container_id"
|
||||||
|
docker rm "$container_id"
|
||||||
|
"${compose[@]}" up -d --no-deps --scale "web=$LOAD_WEB_REPLICAS" web
|
||||||
|
} >>"$output_dir/web-replacements.txt"
|
||||||
|
wait_for_service web "$LOAD_WEB_REPLICAS" true
|
||||||
|
done
|
||||||
|
|
||||||
|
mapfile -t original_worker_ids < <(service_ids worker)
|
||||||
|
|
||||||
|
for container_id in "${original_worker_ids[@]}"; do
|
||||||
|
assert_scope "$container_id" worker
|
||||||
|
{
|
||||||
|
docker stop --time 30 "$container_id"
|
||||||
|
docker rm "$container_id"
|
||||||
|
"${compose[@]}" up -d --no-deps --scale "worker=$LOAD_WORKER_REPLICAS" worker
|
||||||
|
} >>"$output_dir/worker-replacements.txt"
|
||||||
|
wait_for_service worker "$LOAD_WORKER_REPLICAS" false
|
||||||
|
done
|
||||||
|
|
||||||
|
wait_for_cluster >"$output_dir/cluster-after-replacements.txt"
|
||||||
|
|
||||||
|
COMPOSE_PROJECT_NAME=$LOAD_PROJECT "$ROOT/scripts/verify-realtime-cluster.sh" compose \
|
||||||
|
>"$output_dir/pubsub-after-replacements.txt"
|
||||||
|
|
||||||
|
worker_target=$(service_ids worker | head -n 1)
|
||||||
|
assert_scope "$worker_target" worker
|
||||||
|
|
||||||
|
docker exec "$worker_target" /app/bin/who_need_help rpc \
|
||||||
|
"run_id = \"$LABEL\"
|
||||||
|
{:ok, job} =
|
||||||
|
%{
|
||||||
|
\"confirmation\" => \"isolated-local-resilience-probe\",
|
||||||
|
\"run_id\" => run_id
|
||||||
|
}
|
||||||
|
|> WhoNeedHelp.Workers.LocalRetryProbe.new()
|
||||||
|
|> Oban.insert()
|
||||||
|
IO.puts(job.id)" >"$output_dir/retry-job-insert.txt"
|
||||||
|
retry_job_inserted=true
|
||||||
|
|
||||||
|
retry_deadline=$((SECONDS + LOAD_RESILIENCE_RECOVERY_TIMEOUT_SECONDS))
|
||||||
|
|
||||||
|
while ((SECONDS < retry_deadline)); do
|
||||||
|
# The run label is restricted above and passed as a psql value.
|
||||||
|
# shellcheck disable=SC2016
|
||||||
|
"${compose[@]}" exec -T db sh -c \
|
||||||
|
'psql --no-psqlrc --tuples-only --no-align --set ON_ERROR_STOP=1 \
|
||||||
|
--set run_id="$1" --username "$POSTGRES_USER" --dbname "$POSTGRES_DB"' \
|
||||||
|
sh "$LABEL" >"$output_dir/retry-job.json" <<'SQL'
|
||||||
|
SELECT json_build_object(
|
||||||
|
'id', id,
|
||||||
|
'state', state,
|
||||||
|
'attempt', attempt,
|
||||||
|
'max_attempts', max_attempts,
|
||||||
|
'error_count', cardinality(errors),
|
||||||
|
'worker', worker
|
||||||
|
)
|
||||||
|
FROM oban_jobs
|
||||||
|
WHERE worker = 'WhoNeedHelp.Workers.LocalRetryProbe'
|
||||||
|
AND args->>'run_id' = :'run_id'
|
||||||
|
ORDER BY id DESC
|
||||||
|
LIMIT 1;
|
||||||
|
SQL
|
||||||
|
|
||||||
|
if jq -e '
|
||||||
|
.state == "completed" and
|
||||||
|
.attempt == 2 and
|
||||||
|
.max_attempts == 2 and
|
||||||
|
.error_count == 1 and
|
||||||
|
.worker == "WhoNeedHelp.Workers.LocalRetryProbe"
|
||||||
|
' "$output_dir/retry-job.json" >/dev/null 2>&1; then
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
|
||||||
|
sleep 1
|
||||||
|
done
|
||||||
|
|
||||||
|
if ! jq -e '
|
||||||
|
.state == "completed" and
|
||||||
|
.attempt == 2 and
|
||||||
|
.max_attempts == 2 and
|
||||||
|
.error_count == 1 and
|
||||||
|
.worker == "WhoNeedHelp.Workers.LocalRetryProbe"
|
||||||
|
' "$output_dir/retry-job.json" >/dev/null; then
|
||||||
|
echo "The isolated Oban retry probe did not complete its second attempt." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
delete_retry_job "$output_dir/retry-job-cleanup.txt"
|
||||||
|
|
||||||
|
if [[ "$(tail -n 1 "$output_dir/retry-job-cleanup.txt")" != "1" ]]; then
|
||||||
|
echo "The exact resilience probe job was not removed." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
retry_job_inserted=false
|
||||||
|
|
||||||
|
stop_probe
|
||||||
|
probe_pid=
|
||||||
|
|
||||||
|
jq -s '{
|
||||||
|
samples: length,
|
||||||
|
failures: (map(select(.status != "200")) | length),
|
||||||
|
nodes: (map(.body | fromjson? | .node) | map(select(. != null)) | unique)
|
||||||
|
}' "$probe_log" >"$output_dir/readiness-summary.json"
|
||||||
|
|
||||||
|
if ! jq -e '.samples > 0 and .failures == 0 and (.nodes | length) >= 2' \
|
||||||
|
"$output_dir/readiness-summary.json" >/dev/null; then
|
||||||
|
echo "Readiness was unavailable or did not reach multiple web replicas." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
"${compose[@]}" ps -a >"$output_dir/compose-after.txt"
|
||||||
|
"${compose[@]}" logs --since "$run_started_at" proxy web worker \
|
||||||
|
>"$output_dir/application.log" 2>&1
|
||||||
|
|
||||||
|
trap - EXIT HUP INT TERM
|
||||||
|
printf 'Resilience evidence: %s\n' "$output_dir"
|
||||||
24
test/who_need_help/workers/local_retry_probe_test.exs
Normal file
24
test/who_need_help/workers/local_retry_probe_test.exs
Normal file
|
|
@ -0,0 +1,24 @@
|
||||||
|
defmodule WhoNeedHelp.Workers.LocalRetryProbeTest do
|
||||||
|
use ExUnit.Case, async: true
|
||||||
|
use Oban.Testing, repo: WhoNeedHelp.Repo
|
||||||
|
|
||||||
|
alias WhoNeedHelp.Workers.LocalRetryProbe
|
||||||
|
|
||||||
|
@args %{
|
||||||
|
"confirmation" => "isolated-local-resilience-probe",
|
||||||
|
"run_id" => "unit-probe"
|
||||||
|
}
|
||||||
|
|
||||||
|
test "fails the first attempt and succeeds on retry" do
|
||||||
|
assert {:error, :expected_first_attempt_failure} =
|
||||||
|
perform_job(LocalRetryProbe, @args, attempt: 1)
|
||||||
|
|
||||||
|
assert :ok = perform_job(LocalRetryProbe, @args, attempt: 2)
|
||||||
|
assert 1 = LocalRetryProbe.backoff(%Oban.Job{attempt: 1})
|
||||||
|
end
|
||||||
|
|
||||||
|
test "cancels jobs without the local confirmation" do
|
||||||
|
assert {:cancel, :invalid_local_resilience_probe} =
|
||||||
|
perform_job(LocalRetryProbe, %{"run_id" => "unit-probe"}, attempt: 1)
|
||||||
|
end
|
||||||
|
end
|
||||||
Loading…
Reference in New Issue
Block a user