From 15c976213a4f3880ec3597e0f046bb8a3a7feffb Mon Sep 17 00:00:00 2001 From: SimpleTest Date: Tue, 25 Aug 2026 15:27:42 +0300 Subject: [PATCH] Harden backup readiness and external monitoring --- docs/operations.md | 18 +++++ docs/public-launch-checklist.md | 5 ++ .../install-production-external-monitor.sh | 15 ++++ scripts/lib/postgres-readiness.sh | 68 ++++++++++++++++ scripts/production-offsite-backup.sh | 34 ++------ scripts/quality.sh | 3 + ...nstall_production_external_monitor_test.py | 24 +++++- ...roduction_offsite_backup_readiness_test.sh | 79 +++++++++++++++++++ 8 files changed, 219 insertions(+), 27 deletions(-) create mode 100644 scripts/lib/postgres-readiness.sh create mode 100644 test/scripts/production_offsite_backup_readiness_test.sh diff --git a/docs/operations.md b/docs/operations.md index 0d3ed77..3a63090 100644 --- a/docs/operations.md +++ b/docs/operations.md @@ -773,6 +773,7 @@ Install it from the production SMTP configuration without printing the SMTP credential: ```bash +ssh buyvm-maya 'loginctl show-user simple --property=Linger --value' ./scripts/install-production-external-monitor.sh ssh buyvm-maya \ 'systemctl --user status who-need-help-production-monitor.timer --no-pager' @@ -780,6 +781,23 @@ ssh buyvm-maya \ 'journalctl --user -u who-need-help-production-monitor.service --no-pager' ``` +The first command must print `yes`. A user timer is not independent of SSH +sessions merely because it is enabled: the external host must keep that user's +systemd manager running after logout and start it at boot. The installer checks +this before copying configuration or unit files and refuses installation when +lingering is not enabled. An administrator on the external monitor host enables +it explicitly: + +```bash +sudo loginctl enable-linger simple +loginctl show-user simple --property=Linger --value +``` + +This affects every enabled user unit for `simple`, not only the Who Need Help +monitor. To reverse it, first disable and remove the monitor user units, verify +that no other enabled user unit requires persistence, and only then run +`sudo loginctl disable-linger simple`. + The default schedule is once per minute with separate three-second readiness and metrics timeouts. Those defaults match the current application container health timeout and were installed only after public readiness requests from diff --git a/docs/public-launch-checklist.md b/docs/public-launch-checklist.md index 8a5f6d7..b71ff52 100644 --- a/docs/public-launch-checklist.md +++ b/docs/public-launch-checklist.md @@ -155,10 +155,15 @@ source database: ./scripts/production-offsite-backup.sh plan ./scripts/production-offsite-backup.sh run systemctl --user status who-need-help-production-backup.service --no-pager +ssh buyvm-maya \ + 'loginctl show-user simple --property=Linger --value' ssh buyvm-maya \ 'systemctl --user status who-need-help-production-monitor.timer --no-pager' ``` +The linger check must print `yes`; otherwise the enabled user timer can depend +on an active login session and is not a persistent external monitor. + Do not mark backup ownership complete merely because this command passes. The operator must still store the Restic key independently and approve retention, RPO, RTO, capacity, and responsible owners. diff --git a/scripts/install-production-external-monitor.sh b/scripts/install-production-external-monitor.sh index 3054197..84f938d 100755 --- a/scripts/install-production-external-monitor.sh +++ b/scripts/install-production-external-monitor.sh @@ -88,6 +88,20 @@ if [ -n "$backup_max_age" ]; then fi fi +if ! monitor_identity=$(ssh -o BatchMode=yes "$monitor_target" \ + 'printf "%s:%s\n" "$(id -un)" "$(loginctl show-user "$(id -un)" --property=Linger --value)"'); then + echo "Unable to verify systemd user lingering on the external monitor host." >&2 + exit 2 +fi +monitor_user=${monitor_identity%%:*} +monitor_linger=${monitor_identity#*:} +if [ -z "$monitor_user" ] || [ "$monitor_linger" != yes ]; then + echo "External monitor installation requires systemd user lingering for '$monitor_user' on '$monitor_target'." >&2 + echo "An administrator must run: sudo loginctl enable-linger '$monitor_user'" >&2 + echo "Verify with: loginctl show-user '$monitor_user' --property=Linger --value" >&2 + exit 2 +fi + work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX") config="$work_dir/monitor.json" service="$work_dir/who-need-help-production-monitor.service" @@ -221,6 +235,7 @@ ssh -o BatchMode=yes "$monitor_target" \ "chmod 700 '$remote_root/production-external-monitor.py'; chmod 600 '$remote_config' /home/simple/.config/systemd/user/who-need-help-production-monitor.service /home/simple/.config/systemd/user/who-need-help-production-monitor.timer; systemctl --user daemon-reload; systemctl --user start who-need-help-production-monitor.service; systemctl --user enable --now who-need-help-production-monitor.timer" printf 'External monitor installed on %s (%s).\n' "$monitor_target" "$monitor_host" +printf 'Persistent systemd user manager verified for %s (linger=yes).\n' "$monitor_user" printf 'Health URL: %s\n' "$monitor_url" printf 'Metrics URL: %s\n' "$metrics_url" printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \ diff --git a/scripts/lib/postgres-readiness.sh b/scripts/lib/postgres-readiness.sh new file mode 100644 index 0000000..90d64cf --- /dev/null +++ b/scripts/lib/postgres-readiness.sh @@ -0,0 +1,68 @@ +#!/usr/bin/env bash + +capture_postgres_readiness_evidence() { + local container=$1 + local evidence_file=$2 + + docker logs "$container" >"$evidence_file" 2>&1 || true +} + +wait_for_postgres_final_ready() { + local container=$1 + local evidence_file=$2 + local max_attempts=$3 + local interval_seconds=$4 + local attempt=0 + local running + local health + + while ! docker logs "$container" 2>&1 | + grep -Fq 'PostgreSQL init process complete; ready for start up.'; do + attempt=$((attempt + 1)) + running=$(docker inspect --format '{{.State.Running}}' "$container" 2>/dev/null || printf missing) + if [[ "$running" != true ]]; then + capture_postgres_readiness_evidence "$container" "$evidence_file" + echo "Isolated restore PostgreSQL exited during initialization." >&2 + return 1 + fi + if ((attempt >= max_attempts)); then + capture_postgres_readiness_evidence "$container" "$evidence_file" + echo "Isolated restore PostgreSQL did not finish initialization." >&2 + return 1 + fi + sleep "$interval_seconds" + done + + attempt=0 + while true; do + # Docker can retain a cached healthy state from the temporary bootstrap + # postmaster while that server shuts down. Query the final server directly + # and proceed only after it can execute SQL. + if docker exec "$container" psql \ + --username postgres \ + --dbname postgres \ + --tuples-only \ + --no-align \ + --command='select 1;' >/dev/null 2>&1; then + return 0 + fi + + attempt=$((attempt + 1)) + running=$(docker inspect --format '{{.State.Running}}' "$container" 2>/dev/null || printf missing) + health=$(docker inspect \ + --format '{{if .State.Health}}{{.State.Health.Status}}{{else}}missing{{end}}' \ + "$container" 2>/dev/null || printf missing) + + if [[ "$running" != true ]]; then + capture_postgres_readiness_evidence "$container" "$evidence_file" + echo "Isolated restore PostgreSQL exited before accepting SQL." >&2 + return 1 + fi + if [[ "$health" == unhealthy ]] || ((attempt >= max_attempts)); then + capture_postgres_readiness_evidence "$container" "$evidence_file" + echo "Isolated restore PostgreSQL did not become SQL-ready." >&2 + return 1 + fi + sleep "$interval_seconds" + done +} diff --git a/scripts/production-offsite-backup.sh b/scripts/production-offsite-backup.sh index 0ea3a2f..8806f42 100755 --- a/scripts/production-offsite-backup.sh +++ b/scripts/production-offsite-backup.sh @@ -3,6 +3,9 @@ set -euo pipefail umask 077 ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd) +# ROOT is resolved from this script at runtime. +# shellcheck disable=SC1091 +source "$ROOT/scripts/lib/postgres-readiness.sh" action=${1:-plan} config=${2:-"$ROOT/tmp/production-operations/backup.env"} restic="$ROOT/.tools/restic/restic" @@ -219,7 +222,6 @@ restore_snapshot() { local selected_snapshot=$1 local dump="$work_dir/production.dump" local password - local health local tables local migrations local postgis @@ -250,31 +252,11 @@ restore_snapshot() { "$restore_image" >/dev/null restore_started=true - # A freshly initialized postgres/postgis container briefly accepts - # connections through its temporary bootstrap server. Wait until the image - # has completed that bootstrap and started the final server before restoring. - while ! docker logs "$restore_container" 2>&1 | - grep -Fq 'PostgreSQL init process complete; ready for start up.'; do - if [[ "$(docker inspect --format '{{.State.Running}}' "$restore_container")" != true ]]; then - docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true - echo "Isolated restore PostgreSQL exited during initialization." >&2 - return 1 - fi - sleep 1 - done - - while true; do - health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{else}}missing{{end}}' "$restore_container") - case "$health" in - healthy) break ;; - unhealthy) - docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true - echo "Isolated restore PostgreSQL became unhealthy." >&2 - return 1 - ;; - esac - sleep 1 - done + wait_for_postgres_final_ready \ + "$restore_container" \ + "$evidence_dir/restore-postgres.log" \ + 120 \ + 1 # The PostGIS image initializes its requested database with PostGIS already # installed. Restore into a database created from template0 instead, so the diff --git a/scripts/quality.sh b/scripts/quality.sh index f728608..60c395f 100755 --- a/scripts/quality.sh +++ b/scripts/quality.sh @@ -128,6 +128,9 @@ docker run --rm \ "$SHELLCHECK_IMAGE" \ $(find scripts -type f -name '*.sh' -print | sort) +echo "Checking production restore PostgreSQL readiness" +bash test/scripts/production_offsite_backup_readiness_test.sh + echo "Checking the production read-only load safety boundary" ./scripts/production-readonly-load-drill.sh diff --git a/test/scripts/install_production_external_monitor_test.py b/test/scripts/install_production_external_monitor_test.py index 48522a9..0c7bf42 100644 --- a/test/scripts/install_production_external_monitor_test.py +++ b/test/scripts/install_production_external_monitor_test.py @@ -74,6 +74,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase): textwrap.dedent( """\ #!/usr/bin/env python3 + import os import shlex import subprocess import sys @@ -83,6 +84,9 @@ class InstallProductionExternalMonitorTest(unittest.TestCase): raise SystemExit(0) command = sys.argv[-1] + if "loginctl show-user" in command: + print(f"monitor-user:{os.environ.get('FAKE_MONITOR_LINGER', 'yes')}") + raise SystemExit(0) if command.startswith("python3 - "): result = subprocess.run( shlex.split(command), @@ -116,7 +120,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase): ), ) - def run_installer(self, *, override, backup_max_age=None): + def run_installer(self, *, override, backup_max_age=None, linger="yes"): self.capture.unlink(missing_ok=True) env = os.environ.copy() env.update( @@ -127,6 +131,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase): "PRODUCTION_ENV_PATH": str(self.production_env), "FAKE_MONITOR_CAPTURE": str(self.capture), "MONITOR_INSTALL_WORK_ROOT": str(self.work_root), + "FAKE_MONITOR_LINGER": linger, } ) if override: @@ -190,6 +195,23 @@ class InstallProductionExternalMonitorTest(unittest.TestCase): self.assertFalse(self.capture.exists()) self.assertIn("must be a positive integer", result.stderr) + def test_linger_disabled_fails_before_external_changes(self): + result = self.run_installer(override=True, linger="no") + + self.assertEqual(result.returncode, 2) + self.assertFalse(self.capture.exists()) + self.assertIn("requires systemd user lingering", result.stderr) + self.assertIn("sudo loginctl enable-linger 'monitor-user'", result.stderr) + + def test_success_reports_verified_persistent_user_manager(self): + result = self.run_installer(override=True, linger="yes") + + self.assertEqual(result.returncode, 0, result.stderr) + self.assertIn( + "Persistent systemd user manager verified for monitor-user (linger=yes).", + result.stdout, + ) + if __name__ == "__main__": unittest.main() diff --git a/test/scripts/production_offsite_backup_readiness_test.sh b/test/scripts/production_offsite_backup_readiness_test.sh new file mode 100644 index 0000000..da64211 --- /dev/null +++ b/test/scripts/production_offsite_backup_readiness_test.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +set -euo pipefail + +ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/../.." && pwd) + +# ROOT is resolved from this test at runtime. +# shellcheck disable=SC1091 +source "$ROOT/scripts/lib/postgres-readiness.sh" + +mode=eventual_ready +sql_attempts=0 + +docker() { + case "$1" in + logs) + printf '%s\n' 'PostgreSQL init process complete; ready for start up.' + ;; + inspect) + case "$3" in + *State.Running*) printf '%s\n' true ;; + *State.Health*) + if [[ "$mode" == unhealthy ]]; then + printf '%s\n' unhealthy + else + printf '%s\n' healthy + fi + ;; + *) return 2 ;; + esac + ;; + exec) + sql_attempts=$((sql_attempts + 1)) + if [[ "$mode" == unhealthy ]] || ((sql_attempts == 1)); then + return 1 + fi + printf '%s\n' 1 + ;; + *) + return 2 + ;; + esac +} + +sleep() { + : +} + +evidence=$(mktemp) +trap 'rm -f "$evidence"' EXIT + +wait_for_postgres_final_ready restore-probe "$evidence" 3 0 + +if ((sql_attempts != 2)); then + printf 'expected two direct SQL readiness attempts, observed %s\n' "$sql_attempts" >&2 + exit 1 +fi + +if [[ -s "$evidence" ]]; then + echo 'success path unexpectedly wrote failure evidence' >&2 + exit 1 +fi + +mode=unhealthy +sql_attempts=0 +: >"$evidence" + +if wait_for_postgres_final_ready restore-probe "$evidence" 3 0; then + echo 'unhealthy final PostgreSQL unexpectedly passed readiness' >&2 + exit 1 +fi + +if ((sql_attempts != 1)); then + printf 'expected one SQL attempt before unhealthy failure, observed %s\n' "$sql_attempts" >&2 + exit 1 +fi + +grep -Fq 'PostgreSQL init process complete; ready for start up.' "$evidence" + +echo 'Production restore readiness regression passed.'