#!/usr/bin/env bash set -euo pipefail umask 077 ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd) action=${1:-plan} config=${2:-"$ROOT/tmp/production-operations/backup.env"} restic="$ROOT/.tools/restic/restic" restore_image='postgis/postgis:18-3.6-alpine@sha256:05d68c7f0f19b9aa0bf7c4a2049b2e8b38b44a63116392b95726a4c913766cf6' case "$action" in plan | init | run | check | restore-drill) ;; *) echo "Usage: $0 [plan|init|run|check|restore-drill] [CONFIG]" >&2 exit 2 ;; esac if [[ "$config" != /* ]]; then config="$ROOT/$config" fi if [[ ! -f "$config" ]]; then echo "Operations configuration does not exist: $config" >&2 exit 2 fi if [[ "$(stat -c '%a' "$config")" != 600 ]]; then echo "Operations configuration must have mode 0600: $config" >&2 exit 2 fi # shellcheck source=/dev/null source "$config" : "${PRODUCTION_SSH_TARGET:?Set PRODUCTION_SSH_TARGET}" : "${PRODUCTION_REMOTE_ROOT:?Set PRODUCTION_REMOTE_ROOT}" : "${PRODUCTION_EXPECTED_ENVIRONMENT:?Set PRODUCTION_EXPECTED_ENVIRONMENT}" : "${OFFSITE_RESTIC_REPOSITORY:?Set OFFSITE_RESTIC_REPOSITORY}" : "${OFFSITE_RESTIC_PASSWORD_FILE:?Set OFFSITE_RESTIC_PASSWORD_FILE}" : "${OFFSITE_RESTIC_HOST:?Set OFFSITE_RESTIC_HOST}" : "${OFFSITE_RESTIC_TAG:?Set OFFSITE_RESTIC_TAG}" OFFSITE_BACKUP_HEARTBEAT_PATH=${OFFSITE_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json} if [[ "$PRODUCTION_EXPECTED_ENVIRONMENT" != production ]]; then echo "The source environment must be exactly production." >&2 exit 2 fi case "$PRODUCTION_REMOTE_ROOT" in /srv/who_need_help-production) ;; *) echo "The production root is outside the reviewed deployment path." >&2 exit 2 ;; esac case "$OFFSITE_RESTIC_REPOSITORY" in sftp:*) ;; *) echo "This workflow currently accepts only an encrypted Restic SFTP repository." >&2 exit 2 ;; esac if [[ "$OFFSITE_BACKUP_HEARTBEAT_PATH" != /home/simple/.local/state/who-need-help/production-backup.json ]]; then echo "The backup heartbeat path is outside the reviewed independent-monitor state path." >&2 exit 2 fi if [[ ! -f "$OFFSITE_RESTIC_PASSWORD_FILE" ]] || [[ "$(stat -c '%a' "$OFFSITE_RESTIC_PASSWORD_FILE")" != 600 ]]; then echo "The Restic password file must exist with mode 0600." >&2 exit 2 fi if [[ ! -x "$restic" ]]; then echo "Pinned Restic is unavailable. Run ./scripts/bootstrap-restic.sh first." >&2 exit 2 fi for command in docker flock grep jq openssl pg_restore scp sha256sum ssh; do command -v "$command" >/dev/null 2>&1 || { echo "Required command is unavailable: $command" >&2 exit 2 } done export RESTIC_REPOSITORY="$OFFSITE_RESTIC_REPOSITORY" export RESTIC_PASSWORD_FILE="$OFFSITE_RESTIC_PASSWORD_FILE" export RESTIC_CACHE_DIR="$ROOT/tmp/production-operations/restic-cache" mkdir -p "$RESTIC_CACHE_DIR" chmod 700 "$ROOT/tmp" "$ROOT/tmp/production-operations" "$RESTIC_CACHE_DIR" source_host=$(ssh -G "$PRODUCTION_SSH_TARGET" | awk '$1 == "hostname" {print $2; exit}') repository_alias=${OFFSITE_RESTIC_REPOSITORY#sftp:} repository_alias=${repository_alias%%:*} repository_host=$(ssh -G "$repository_alias" | awk '$1 == "hostname" {print $2; exit}') if [[ -z "$source_host" || -z "$repository_host" ]] || [[ "$source_host" == "$repository_host" ]]; then echo "The repository host must resolve and differ from production." >&2 exit 2 fi check_source() { ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \ "bash -s -- '$PRODUCTION_REMOTE_ROOT/.env' --check-only '$PRODUCTION_EXPECTED_ENVIRONMENT'" \ <"$ROOT/scripts/backup-external-postgres.sh" } repo_initialized() { "$restic" snapshots --json >/dev/null 2>&1 } printf 'Production SSH target: %s (%s)\n' "$PRODUCTION_SSH_TARGET" "$source_host" printf 'Production root: %s\n' "$PRODUCTION_REMOTE_ROOT" printf 'Encrypted repository: %s (%s)\n' "$OFFSITE_RESTIC_REPOSITORY" "$repository_host" printf 'Restic: %s\n' "$($restic version)" if [[ "$action" == plan ]]; then check_source if repo_initialized; then echo "Encrypted repository opened successfully." else echo "Encrypted repository is not initialized or cannot be opened." fi echo "Plan completed without changing production or the repository." exit 0 fi if [[ "$action" == init ]]; then confirmation="$OFFSITE_RESTIC_REPOSITORY" if [[ "${WNH_OFFSITE_BACKUP_INIT_CONFIRM:-}" != "$confirmation" ]]; then echo "Repository initialization requires exact confirmation:" >&2 echo "WNH_OFFSITE_BACKUP_INIT_CONFIRM=$confirmation $0 init '$config'" >&2 exit 2 fi check_source if repo_initialized; then echo "Encrypted repository is already initialized; no change was made." exit 0 fi "$restic" init "$restic" check echo "Encrypted off-server Restic repository initialized and opened successfully." exit 0 fi repo_initialized || { echo "Encrypted repository cannot be opened. Initialize it first." >&2 exit 2 } if [[ "$action" == check ]]; then "$restic" check --read-data echo "Encrypted repository full-data check passed." exit 0 fi if [[ "$action" == restore-drill ]]; then snapshot_id=$($restic snapshots --json --host "$OFFSITE_RESTIC_HOST" \ --tag "$OFFSITE_RESTIC_TAG" --latest 1 | jq -r '.[0].short_id // empty') if [[ -z "$snapshot_id" ]]; then echo "No production backup snapshot is available for a restore drill." >&2 exit 1 fi else snapshot_id= fi lock_file="$ROOT/tmp/production-operations/backup.lock" exec 9>"$lock_file" flock --nonblock 9 || { echo "Another production backup or restore drill is already running." >&2 exit 1 } run_id=$(date -u +%Y%m%dT%H%M%SZ)-$$ work_dir=$(mktemp -d "$ROOT/tmp/production-operations/run-$run_id.XXXXXX") evidence_dir="$ROOT/output/production-operations/$run_id" remote_dump="$PRODUCTION_REMOTE_ROOT/output/backups/production/scheduled-$run_id.dump" remote_created=false restore_container="wnh-production-restore-$run_id" restore_volume="wnh_production_restore_${run_id//[^a-zA-Z0-9]/_}" restore_started=false restore_volume_created=false heartbeat_staged=false heartbeat_remote_tmp="$OFFSITE_BACKUP_HEARTBEAT_PATH.tmp-$run_id" mkdir -p "$evidence_dir" chmod 700 "$ROOT/output" "$ROOT/output/production-operations" "$evidence_dir" "$work_dir" cleanup() { trap - EXIT HUP INT TERM if [[ "$restore_started" == true ]]; then docker rm --force "$restore_container" >/dev/null 2>&1 || true fi if [[ "$restore_volume_created" == true ]]; then docker volume rm "$restore_volume" >/dev/null 2>&1 || true fi if [[ "$remote_created" == true ]]; then ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \ "rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" \ >/dev/null 2>&1 || true fi if [[ "$heartbeat_staged" == true ]]; then ssh -o BatchMode=yes "$repository_alias" \ "rm -f -- '$heartbeat_remote_tmp'" >/dev/null 2>&1 || true fi rm -rf "$work_dir" } trap cleanup EXIT HUP INT TERM restore_snapshot() { local selected_snapshot=$1 local dump="$work_dir/production.dump" local password local health local tables local migrations local postgis "$restic" dump "$selected_snapshot" /production.dump >"$dump" [[ -s "$dump" ]] || { echo "Restic restored an empty database dump." >&2 return 1 } pg_restore --list "$dump" >"$evidence_dir/pg-restore-catalog.txt" if ! docker image inspect "$restore_image" >/dev/null 2>&1; then docker pull "$restore_image" >"$evidence_dir/postgis-pull.txt" fi password=$(openssl rand -base64 36 | tr -d '\n') docker volume create "$restore_volume" >/dev/null restore_volume_created=true docker run --detach \ --name "$restore_container" \ --env POSTGRES_PASSWORD="$password" \ --env POSTGRES_DB=postgres \ --volume "$restore_volume:/var/lib/postgresql" \ --health-cmd='pg_isready --username postgres --dbname postgres' \ --health-interval=1s \ --health-timeout=2s \ --health-retries=120 \ "$restore_image" >/dev/null restore_started=true # A freshly initialized postgres/postgis container briefly accepts # connections through its temporary bootstrap server. Wait until the image # has completed that bootstrap and started the final server before restoring. while ! docker logs "$restore_container" 2>&1 | grep -Fq 'PostgreSQL init process complete; ready for start up.'; do if [[ "$(docker inspect --format '{{.State.Running}}' "$restore_container")" != true ]]; then docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true echo "Isolated restore PostgreSQL exited during initialization." >&2 return 1 fi sleep 1 done while true; do health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{else}}missing{{end}}' "$restore_container") case "$health" in healthy) break ;; unhealthy) docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true echo "Isolated restore PostgreSQL became unhealthy." >&2 return 1 ;; esac sleep 1 done # The PostGIS image initializes its requested database with PostGIS already # installed. Restore into a database created from template0 instead, so the # archive remains responsible for recreating its own extensions and schema. docker exec "$restore_container" createdb \ --username postgres \ --template template0 \ restore_check if ! docker exec --interactive "$restore_container" pg_restore \ --username postgres \ --dbname restore_check \ --exit-on-error \ --no-owner \ --no-privileges <"$dump"; then docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true docker inspect "$restore_container" >"$evidence_dir/restore-container.json" 2>&1 || true echo "Isolated restore failed; container evidence was retained." >&2 return 1 fi tables=$(docker exec "$restore_container" psql \ --username postgres --dbname restore_check --tuples-only --no-align \ --command="select count(*) from pg_tables where schemaname='public' and tablename <> 'spatial_ref_sys';") migrations=$(docker exec "$restore_container" psql \ --username postgres --dbname restore_check --tuples-only --no-align \ --command='select count(*) from schema_migrations;') postgis=$(docker exec "$restore_container" psql \ --username postgres --dbname restore_check --tuples-only --no-align \ --command='select PostGIS_Lib_Version();') if [[ ! "$tables" =~ ^[1-9][0-9]*$ ]] || [[ ! "$migrations" =~ ^[1-9][0-9]*$ ]]; then echo "The isolated restore does not contain the expected application schema." >&2 return 1 fi { printf 'snapshot_id=%s\n' "$selected_snapshot" printf 'application_tables=%s\n' "$tables" printf 'schema_migrations=%s\n' "$migrations" printf 'postgis_version=%s\n' "$postgis" printf 'verified_at=%s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" } >"$evidence_dir/restore-drill.txt" chmod 600 "$evidence_dir/restore-drill.txt" "$evidence_dir/pg-restore-catalog.txt" docker rm --force "$restore_container" >/dev/null restore_started=false docker volume rm "$restore_volume" >/dev/null restore_volume_created=false rm -f "$dump" } if [[ "$action" == restore-drill ]]; then restore_snapshot "$snapshot_id" echo "Isolated restore drill passed for snapshot: $snapshot_id" printf 'Non-secret evidence: %s\n' "$evidence_dir" exit 0 fi check_source ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \ "bash -s -- '$PRODUCTION_REMOTE_ROOT/.env' '$remote_dump' '$PRODUCTION_EXPECTED_ENVIRONMENT'" \ <"$ROOT/scripts/backup-external-postgres.sh" remote_created=true scp -p \ "$PRODUCTION_SSH_TARGET:$remote_dump" \ "$PRODUCTION_SSH_TARGET:$remote_dump.sha256" \ "$PRODUCTION_SSH_TARGET:$remote_dump.metadata" \ "$work_dir/" dump="$work_dir/$(basename -- "$remote_dump")" checksum="$dump.sha256" metadata="$dump.metadata" ( cd "$work_dir" sha256sum --check "$(basename -- "$checksum")" ) >"$evidence_dir/source-checksum.txt" pg_restore --list "$dump" >"$evidence_dir/source-catalog.txt" cp "$checksum" "$metadata" "$evidence_dir/" chmod 600 "$evidence_dir"/* run_tag="$OFFSITE_RESTIC_TAG-$run_id" backup_json="$evidence_dir/restic-backup.jsonl" "$restic" backup \ --json \ --host "$OFFSITE_RESTIC_HOST" \ --tag "$OFFSITE_RESTIC_TAG" \ --tag "$run_tag" \ --stdin-filename production.dump \ --stdin-from-command -- cat "$dump" >"$backup_json" snapshot_id=$(jq -r 'select(.message_type == "summary") | .snapshot_id // empty' "$backup_json" | tail -n 1) if [[ -z "$snapshot_id" ]]; then echo "Restic did not report a completed snapshot." >&2 exit 1 fi snapshot_matches=$($restic snapshots --json --tag "$run_tag" | jq 'length') if [[ "$snapshot_matches" != 1 ]]; then echo "The run tag did not resolve to exactly one snapshot." >&2 exit 1 fi "$restic" check --read-data >"$evidence_dir/restic-check.txt" restore_snapshot "$snapshot_id" printf '%s\n' "$snapshot_id" >"$evidence_dir/snapshot-id.txt" chmod 600 "$evidence_dir"/* ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \ "rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" remote_created=false heartbeat="$work_dir/production-backup-heartbeat.json" jq -n \ --arg snapshot_id "$snapshot_id" \ --arg verified_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ '{snapshot_id: $snapshot_id, verified_at: $verified_at, restore_verified: true}' \ >"$heartbeat" chmod 600 "$heartbeat" ssh -o BatchMode=yes "$repository_alias" \ "install -d -m 700 '$(dirname -- "$OFFSITE_BACKUP_HEARTBEAT_PATH")'" heartbeat_staged=true scp -q "$heartbeat" "$repository_alias:$heartbeat_remote_tmp" ssh -o BatchMode=yes "$repository_alias" \ "chmod 600 '$heartbeat_remote_tmp' && mv -f -- '$heartbeat_remote_tmp' '$OFFSITE_BACKUP_HEARTBEAT_PATH'" heartbeat_staged=false echo "Encrypted off-server backup and isolated restore drill passed." printf 'Snapshot: %s\n' "$snapshot_id" printf 'Non-secret evidence: %s\n' "$evidence_dir"