Harden backup readiness and external monitoring

This commit is contained in:
SimpleTest 2026-08-25 15:27:42 +03:00
parent f15f5578e0
commit 15c976213a
8 changed files with 219 additions and 27 deletions

View File

@ -773,6 +773,7 @@ Install it from the production SMTP configuration without printing the SMTP
credential: credential:
```bash ```bash
ssh buyvm-maya 'loginctl show-user simple --property=Linger --value'
./scripts/install-production-external-monitor.sh ./scripts/install-production-external-monitor.sh
ssh buyvm-maya \ ssh buyvm-maya \
'systemctl --user status who-need-help-production-monitor.timer --no-pager' 'systemctl --user status who-need-help-production-monitor.timer --no-pager'
@ -780,6 +781,23 @@ ssh buyvm-maya \
'journalctl --user -u who-need-help-production-monitor.service --no-pager' 'journalctl --user -u who-need-help-production-monitor.service --no-pager'
``` ```
The first command must print `yes`. A user timer is not independent of SSH
sessions merely because it is enabled: the external host must keep that user's
systemd manager running after logout and start it at boot. The installer checks
this before copying configuration or unit files and refuses installation when
lingering is not enabled. An administrator on the external monitor host enables
it explicitly:
```bash
sudo loginctl enable-linger simple
loginctl show-user simple --property=Linger --value
```
This affects every enabled user unit for `simple`, not only the Who Need Help
monitor. To reverse it, first disable and remove the monitor user units, verify
that no other enabled user unit requires persistence, and only then run
`sudo loginctl disable-linger simple`.
The default schedule is once per minute with separate three-second readiness The default schedule is once per minute with separate three-second readiness
and metrics timeouts. Those defaults match the current application container and metrics timeouts. Those defaults match the current application container
health timeout and were installed only after public readiness requests from health timeout and were installed only after public readiness requests from

View File

@ -155,10 +155,15 @@ source database:
./scripts/production-offsite-backup.sh plan ./scripts/production-offsite-backup.sh plan
./scripts/production-offsite-backup.sh run ./scripts/production-offsite-backup.sh run
systemctl --user status who-need-help-production-backup.service --no-pager systemctl --user status who-need-help-production-backup.service --no-pager
ssh buyvm-maya \
'loginctl show-user simple --property=Linger --value'
ssh buyvm-maya \ ssh buyvm-maya \
'systemctl --user status who-need-help-production-monitor.timer --no-pager' 'systemctl --user status who-need-help-production-monitor.timer --no-pager'
``` ```
The linger check must print `yes`; otherwise the enabled user timer can depend
on an active login session and is not a persistent external monitor.
Do not mark backup ownership complete merely because this command passes. The Do not mark backup ownership complete merely because this command passes. The
operator must still store the Restic key independently and approve retention, operator must still store the Restic key independently and approve retention,
RPO, RTO, capacity, and responsible owners. RPO, RTO, capacity, and responsible owners.

View File

@ -88,6 +88,20 @@ if [ -n "$backup_max_age" ]; then
fi fi
fi fi
if ! monitor_identity=$(ssh -o BatchMode=yes "$monitor_target" \
'printf "%s:%s\n" "$(id -un)" "$(loginctl show-user "$(id -un)" --property=Linger --value)"'); then
echo "Unable to verify systemd user lingering on the external monitor host." >&2
exit 2
fi
monitor_user=${monitor_identity%%:*}
monitor_linger=${monitor_identity#*:}
if [ -z "$monitor_user" ] || [ "$monitor_linger" != yes ]; then
echo "External monitor installation requires systemd user lingering for '$monitor_user' on '$monitor_target'." >&2
echo "An administrator must run: sudo loginctl enable-linger '$monitor_user'" >&2
echo "Verify with: loginctl show-user '$monitor_user' --property=Linger --value" >&2
exit 2
fi
work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX") work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX")
config="$work_dir/monitor.json" config="$work_dir/monitor.json"
service="$work_dir/who-need-help-production-monitor.service" service="$work_dir/who-need-help-production-monitor.service"
@ -221,6 +235,7 @@ ssh -o BatchMode=yes "$monitor_target" \
"chmod 700 '$remote_root/production-external-monitor.py'; chmod 600 '$remote_config' /home/simple/.config/systemd/user/who-need-help-production-monitor.service /home/simple/.config/systemd/user/who-need-help-production-monitor.timer; systemctl --user daemon-reload; systemctl --user start who-need-help-production-monitor.service; systemctl --user enable --now who-need-help-production-monitor.timer" "chmod 700 '$remote_root/production-external-monitor.py'; chmod 600 '$remote_config' /home/simple/.config/systemd/user/who-need-help-production-monitor.service /home/simple/.config/systemd/user/who-need-help-production-monitor.timer; systemctl --user daemon-reload; systemctl --user start who-need-help-production-monitor.service; systemctl --user enable --now who-need-help-production-monitor.timer"
printf 'External monitor installed on %s (%s).\n' "$monitor_target" "$monitor_host" printf 'External monitor installed on %s (%s).\n' "$monitor_target" "$monitor_host"
printf 'Persistent systemd user manager verified for %s (linger=yes).\n' "$monitor_user"
printf 'Health URL: %s\n' "$monitor_url" printf 'Health URL: %s\n' "$monitor_url"
printf 'Metrics URL: %s\n' "$metrics_url" printf 'Metrics URL: %s\n' "$metrics_url"
printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \ printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \

View File

@ -0,0 +1,68 @@
#!/usr/bin/env bash
capture_postgres_readiness_evidence() {
local container=$1
local evidence_file=$2
docker logs "$container" >"$evidence_file" 2>&1 || true
}
wait_for_postgres_final_ready() {
local container=$1
local evidence_file=$2
local max_attempts=$3
local interval_seconds=$4
local attempt=0
local running
local health
while ! docker logs "$container" 2>&1 |
grep -Fq 'PostgreSQL init process complete; ready for start up.'; do
attempt=$((attempt + 1))
running=$(docker inspect --format '{{.State.Running}}' "$container" 2>/dev/null || printf missing)
if [[ "$running" != true ]]; then
capture_postgres_readiness_evidence "$container" "$evidence_file"
echo "Isolated restore PostgreSQL exited during initialization." >&2
return 1
fi
if ((attempt >= max_attempts)); then
capture_postgres_readiness_evidence "$container" "$evidence_file"
echo "Isolated restore PostgreSQL did not finish initialization." >&2
return 1
fi
sleep "$interval_seconds"
done
attempt=0
while true; do
# Docker can retain a cached healthy state from the temporary bootstrap
# postmaster while that server shuts down. Query the final server directly
# and proceed only after it can execute SQL.
if docker exec "$container" psql \
--username postgres \
--dbname postgres \
--tuples-only \
--no-align \
--command='select 1;' >/dev/null 2>&1; then
return 0
fi
attempt=$((attempt + 1))
running=$(docker inspect --format '{{.State.Running}}' "$container" 2>/dev/null || printf missing)
health=$(docker inspect \
--format '{{if .State.Health}}{{.State.Health.Status}}{{else}}missing{{end}}' \
"$container" 2>/dev/null || printf missing)
if [[ "$running" != true ]]; then
capture_postgres_readiness_evidence "$container" "$evidence_file"
echo "Isolated restore PostgreSQL exited before accepting SQL." >&2
return 1
fi
if [[ "$health" == unhealthy ]] || ((attempt >= max_attempts)); then
capture_postgres_readiness_evidence "$container" "$evidence_file"
echo "Isolated restore PostgreSQL did not become SQL-ready." >&2
return 1
fi
sleep "$interval_seconds"
done
}

View File

@ -3,6 +3,9 @@ set -euo pipefail
umask 077 umask 077
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd) ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)
# ROOT is resolved from this script at runtime.
# shellcheck disable=SC1091
source "$ROOT/scripts/lib/postgres-readiness.sh"
action=${1:-plan} action=${1:-plan}
config=${2:-"$ROOT/tmp/production-operations/backup.env"} config=${2:-"$ROOT/tmp/production-operations/backup.env"}
restic="$ROOT/.tools/restic/restic" restic="$ROOT/.tools/restic/restic"
@ -219,7 +222,6 @@ restore_snapshot() {
local selected_snapshot=$1 local selected_snapshot=$1
local dump="$work_dir/production.dump" local dump="$work_dir/production.dump"
local password local password
local health
local tables local tables
local migrations local migrations
local postgis local postgis
@ -250,31 +252,11 @@ restore_snapshot() {
"$restore_image" >/dev/null "$restore_image" >/dev/null
restore_started=true restore_started=true
# A freshly initialized postgres/postgis container briefly accepts wait_for_postgres_final_ready \
# connections through its temporary bootstrap server. Wait until the image "$restore_container" \
# has completed that bootstrap and started the final server before restoring. "$evidence_dir/restore-postgres.log" \
while ! docker logs "$restore_container" 2>&1 | 120 \
grep -Fq 'PostgreSQL init process complete; ready for start up.'; do 1
if [[ "$(docker inspect --format '{{.State.Running}}' "$restore_container")" != true ]]; then
docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true
echo "Isolated restore PostgreSQL exited during initialization." >&2
return 1
fi
sleep 1
done
while true; do
health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{else}}missing{{end}}' "$restore_container")
case "$health" in
healthy) break ;;
unhealthy)
docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true
echo "Isolated restore PostgreSQL became unhealthy." >&2
return 1
;;
esac
sleep 1
done
# The PostGIS image initializes its requested database with PostGIS already # The PostGIS image initializes its requested database with PostGIS already
# installed. Restore into a database created from template0 instead, so the # installed. Restore into a database created from template0 instead, so the

View File

@ -128,6 +128,9 @@ docker run --rm \
"$SHELLCHECK_IMAGE" \ "$SHELLCHECK_IMAGE" \
$(find scripts -type f -name '*.sh' -print | sort) $(find scripts -type f -name '*.sh' -print | sort)
echo "Checking production restore PostgreSQL readiness"
bash test/scripts/production_offsite_backup_readiness_test.sh
echo "Checking the production read-only load safety boundary" echo "Checking the production read-only load safety boundary"
./scripts/production-readonly-load-drill.sh ./scripts/production-readonly-load-drill.sh

View File

@ -74,6 +74,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
textwrap.dedent( textwrap.dedent(
"""\ """\
#!/usr/bin/env python3 #!/usr/bin/env python3
import os
import shlex import shlex
import subprocess import subprocess
import sys import sys
@ -83,6 +84,9 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
raise SystemExit(0) raise SystemExit(0)
command = sys.argv[-1] command = sys.argv[-1]
if "loginctl show-user" in command:
print(f"monitor-user:{os.environ.get('FAKE_MONITOR_LINGER', 'yes')}")
raise SystemExit(0)
if command.startswith("python3 - "): if command.startswith("python3 - "):
result = subprocess.run( result = subprocess.run(
shlex.split(command), shlex.split(command),
@ -116,7 +120,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
), ),
) )
def run_installer(self, *, override, backup_max_age=None): def run_installer(self, *, override, backup_max_age=None, linger="yes"):
self.capture.unlink(missing_ok=True) self.capture.unlink(missing_ok=True)
env = os.environ.copy() env = os.environ.copy()
env.update( env.update(
@ -127,6 +131,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
"PRODUCTION_ENV_PATH": str(self.production_env), "PRODUCTION_ENV_PATH": str(self.production_env),
"FAKE_MONITOR_CAPTURE": str(self.capture), "FAKE_MONITOR_CAPTURE": str(self.capture),
"MONITOR_INSTALL_WORK_ROOT": str(self.work_root), "MONITOR_INSTALL_WORK_ROOT": str(self.work_root),
"FAKE_MONITOR_LINGER": linger,
} }
) )
if override: if override:
@ -190,6 +195,23 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
self.assertFalse(self.capture.exists()) self.assertFalse(self.capture.exists())
self.assertIn("must be a positive integer", result.stderr) self.assertIn("must be a positive integer", result.stderr)
def test_linger_disabled_fails_before_external_changes(self):
result = self.run_installer(override=True, linger="no")
self.assertEqual(result.returncode, 2)
self.assertFalse(self.capture.exists())
self.assertIn("requires systemd user lingering", result.stderr)
self.assertIn("sudo loginctl enable-linger 'monitor-user'", result.stderr)
def test_success_reports_verified_persistent_user_manager(self):
result = self.run_installer(override=True, linger="yes")
self.assertEqual(result.returncode, 0, result.stderr)
self.assertIn(
"Persistent systemd user manager verified for monitor-user (linger=yes).",
result.stdout,
)
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()

View File

@ -0,0 +1,79 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/../.." && pwd)
# ROOT is resolved from this test at runtime.
# shellcheck disable=SC1091
source "$ROOT/scripts/lib/postgres-readiness.sh"
mode=eventual_ready
sql_attempts=0
docker() {
case "$1" in
logs)
printf '%s\n' 'PostgreSQL init process complete; ready for start up.'
;;
inspect)
case "$3" in
*State.Running*) printf '%s\n' true ;;
*State.Health*)
if [[ "$mode" == unhealthy ]]; then
printf '%s\n' unhealthy
else
printf '%s\n' healthy
fi
;;
*) return 2 ;;
esac
;;
exec)
sql_attempts=$((sql_attempts + 1))
if [[ "$mode" == unhealthy ]] || ((sql_attempts == 1)); then
return 1
fi
printf '%s\n' 1
;;
*)
return 2
;;
esac
}
sleep() {
:
}
evidence=$(mktemp)
trap 'rm -f "$evidence"' EXIT
wait_for_postgres_final_ready restore-probe "$evidence" 3 0
if ((sql_attempts != 2)); then
printf 'expected two direct SQL readiness attempts, observed %s\n' "$sql_attempts" >&2
exit 1
fi
if [[ -s "$evidence" ]]; then
echo 'success path unexpectedly wrote failure evidence' >&2
exit 1
fi
mode=unhealthy
sql_attempts=0
: >"$evidence"
if wait_for_postgres_final_ready restore-probe "$evidence" 3 0; then
echo 'unhealthy final PostgreSQL unexpectedly passed readiness' >&2
exit 1
fi
if ((sql_attempts != 1)); then
printf 'expected one SQL attempt before unhealthy failure, observed %s\n' "$sql_attempts" >&2
exit 1
fi
grep -Fq 'PostgreSQL init process complete; ready for start up.' "$evidence"
echo 'Production restore readiness regression passed.'