Harden backup readiness and external monitoring
This commit is contained in:
parent
f15f5578e0
commit
15c976213a
|
|
@ -773,6 +773,7 @@ Install it from the production SMTP configuration without printing the SMTP
|
|||
credential:
|
||||
|
||||
```bash
|
||||
ssh buyvm-maya 'loginctl show-user simple --property=Linger --value'
|
||||
./scripts/install-production-external-monitor.sh
|
||||
ssh buyvm-maya \
|
||||
'systemctl --user status who-need-help-production-monitor.timer --no-pager'
|
||||
|
|
@ -780,6 +781,23 @@ ssh buyvm-maya \
|
|||
'journalctl --user -u who-need-help-production-monitor.service --no-pager'
|
||||
```
|
||||
|
||||
The first command must print `yes`. A user timer is not independent of SSH
|
||||
sessions merely because it is enabled: the external host must keep that user's
|
||||
systemd manager running after logout and start it at boot. The installer checks
|
||||
this before copying configuration or unit files and refuses installation when
|
||||
lingering is not enabled. An administrator on the external monitor host enables
|
||||
it explicitly:
|
||||
|
||||
```bash
|
||||
sudo loginctl enable-linger simple
|
||||
loginctl show-user simple --property=Linger --value
|
||||
```
|
||||
|
||||
This affects every enabled user unit for `simple`, not only the Who Need Help
|
||||
monitor. To reverse it, first disable and remove the monitor user units, verify
|
||||
that no other enabled user unit requires persistence, and only then run
|
||||
`sudo loginctl disable-linger simple`.
|
||||
|
||||
The default schedule is once per minute with separate three-second readiness
|
||||
and metrics timeouts. Those defaults match the current application container
|
||||
health timeout and were installed only after public readiness requests from
|
||||
|
|
|
|||
|
|
@ -155,10 +155,15 @@ source database:
|
|||
./scripts/production-offsite-backup.sh plan
|
||||
./scripts/production-offsite-backup.sh run
|
||||
systemctl --user status who-need-help-production-backup.service --no-pager
|
||||
ssh buyvm-maya \
|
||||
'loginctl show-user simple --property=Linger --value'
|
||||
ssh buyvm-maya \
|
||||
'systemctl --user status who-need-help-production-monitor.timer --no-pager'
|
||||
```
|
||||
|
||||
The linger check must print `yes`; otherwise the enabled user timer can depend
|
||||
on an active login session and is not a persistent external monitor.
|
||||
|
||||
Do not mark backup ownership complete merely because this command passes. The
|
||||
operator must still store the Restic key independently and approve retention,
|
||||
RPO, RTO, capacity, and responsible owners.
|
||||
|
|
|
|||
|
|
@ -88,6 +88,20 @@ if [ -n "$backup_max_age" ]; then
|
|||
fi
|
||||
fi
|
||||
|
||||
if ! monitor_identity=$(ssh -o BatchMode=yes "$monitor_target" \
|
||||
'printf "%s:%s\n" "$(id -un)" "$(loginctl show-user "$(id -un)" --property=Linger --value)"'); then
|
||||
echo "Unable to verify systemd user lingering on the external monitor host." >&2
|
||||
exit 2
|
||||
fi
|
||||
monitor_user=${monitor_identity%%:*}
|
||||
monitor_linger=${monitor_identity#*:}
|
||||
if [ -z "$monitor_user" ] || [ "$monitor_linger" != yes ]; then
|
||||
echo "External monitor installation requires systemd user lingering for '$monitor_user' on '$monitor_target'." >&2
|
||||
echo "An administrator must run: sudo loginctl enable-linger '$monitor_user'" >&2
|
||||
echo "Verify with: loginctl show-user '$monitor_user' --property=Linger --value" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX")
|
||||
config="$work_dir/monitor.json"
|
||||
service="$work_dir/who-need-help-production-monitor.service"
|
||||
|
|
@ -221,6 +235,7 @@ ssh -o BatchMode=yes "$monitor_target" \
|
|||
"chmod 700 '$remote_root/production-external-monitor.py'; chmod 600 '$remote_config' /home/simple/.config/systemd/user/who-need-help-production-monitor.service /home/simple/.config/systemd/user/who-need-help-production-monitor.timer; systemctl --user daemon-reload; systemctl --user start who-need-help-production-monitor.service; systemctl --user enable --now who-need-help-production-monitor.timer"
|
||||
|
||||
printf 'External monitor installed on %s (%s).\n' "$monitor_target" "$monitor_host"
|
||||
printf 'Persistent systemd user manager verified for %s (linger=yes).\n' "$monitor_user"
|
||||
printf 'Health URL: %s\n' "$monitor_url"
|
||||
printf 'Metrics URL: %s\n' "$metrics_url"
|
||||
printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \
|
||||
|
|
|
|||
68
scripts/lib/postgres-readiness.sh
Normal file
68
scripts/lib/postgres-readiness.sh
Normal file
|
|
@ -0,0 +1,68 @@
|
|||
#!/usr/bin/env bash
|
||||
|
||||
capture_postgres_readiness_evidence() {
|
||||
local container=$1
|
||||
local evidence_file=$2
|
||||
|
||||
docker logs "$container" >"$evidence_file" 2>&1 || true
|
||||
}
|
||||
|
||||
wait_for_postgres_final_ready() {
|
||||
local container=$1
|
||||
local evidence_file=$2
|
||||
local max_attempts=$3
|
||||
local interval_seconds=$4
|
||||
local attempt=0
|
||||
local running
|
||||
local health
|
||||
|
||||
while ! docker logs "$container" 2>&1 |
|
||||
grep -Fq 'PostgreSQL init process complete; ready for start up.'; do
|
||||
attempt=$((attempt + 1))
|
||||
running=$(docker inspect --format '{{.State.Running}}' "$container" 2>/dev/null || printf missing)
|
||||
if [[ "$running" != true ]]; then
|
||||
capture_postgres_readiness_evidence "$container" "$evidence_file"
|
||||
echo "Isolated restore PostgreSQL exited during initialization." >&2
|
||||
return 1
|
||||
fi
|
||||
if ((attempt >= max_attempts)); then
|
||||
capture_postgres_readiness_evidence "$container" "$evidence_file"
|
||||
echo "Isolated restore PostgreSQL did not finish initialization." >&2
|
||||
return 1
|
||||
fi
|
||||
sleep "$interval_seconds"
|
||||
done
|
||||
|
||||
attempt=0
|
||||
while true; do
|
||||
# Docker can retain a cached healthy state from the temporary bootstrap
|
||||
# postmaster while that server shuts down. Query the final server directly
|
||||
# and proceed only after it can execute SQL.
|
||||
if docker exec "$container" psql \
|
||||
--username postgres \
|
||||
--dbname postgres \
|
||||
--tuples-only \
|
||||
--no-align \
|
||||
--command='select 1;' >/dev/null 2>&1; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
attempt=$((attempt + 1))
|
||||
running=$(docker inspect --format '{{.State.Running}}' "$container" 2>/dev/null || printf missing)
|
||||
health=$(docker inspect \
|
||||
--format '{{if .State.Health}}{{.State.Health.Status}}{{else}}missing{{end}}' \
|
||||
"$container" 2>/dev/null || printf missing)
|
||||
|
||||
if [[ "$running" != true ]]; then
|
||||
capture_postgres_readiness_evidence "$container" "$evidence_file"
|
||||
echo "Isolated restore PostgreSQL exited before accepting SQL." >&2
|
||||
return 1
|
||||
fi
|
||||
if [[ "$health" == unhealthy ]] || ((attempt >= max_attempts)); then
|
||||
capture_postgres_readiness_evidence "$container" "$evidence_file"
|
||||
echo "Isolated restore PostgreSQL did not become SQL-ready." >&2
|
||||
return 1
|
||||
fi
|
||||
sleep "$interval_seconds"
|
||||
done
|
||||
}
|
||||
|
|
@ -3,6 +3,9 @@ set -euo pipefail
|
|||
umask 077
|
||||
|
||||
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)
|
||||
# ROOT is resolved from this script at runtime.
|
||||
# shellcheck disable=SC1091
|
||||
source "$ROOT/scripts/lib/postgres-readiness.sh"
|
||||
action=${1:-plan}
|
||||
config=${2:-"$ROOT/tmp/production-operations/backup.env"}
|
||||
restic="$ROOT/.tools/restic/restic"
|
||||
|
|
@ -219,7 +222,6 @@ restore_snapshot() {
|
|||
local selected_snapshot=$1
|
||||
local dump="$work_dir/production.dump"
|
||||
local password
|
||||
local health
|
||||
local tables
|
||||
local migrations
|
||||
local postgis
|
||||
|
|
@ -250,31 +252,11 @@ restore_snapshot() {
|
|||
"$restore_image" >/dev/null
|
||||
restore_started=true
|
||||
|
||||
# A freshly initialized postgres/postgis container briefly accepts
|
||||
# connections through its temporary bootstrap server. Wait until the image
|
||||
# has completed that bootstrap and started the final server before restoring.
|
||||
while ! docker logs "$restore_container" 2>&1 |
|
||||
grep -Fq 'PostgreSQL init process complete; ready for start up.'; do
|
||||
if [[ "$(docker inspect --format '{{.State.Running}}' "$restore_container")" != true ]]; then
|
||||
docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true
|
||||
echo "Isolated restore PostgreSQL exited during initialization." >&2
|
||||
return 1
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
|
||||
while true; do
|
||||
health=$(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{else}}missing{{end}}' "$restore_container")
|
||||
case "$health" in
|
||||
healthy) break ;;
|
||||
unhealthy)
|
||||
docker logs "$restore_container" >"$evidence_dir/restore-postgres.log" 2>&1 || true
|
||||
echo "Isolated restore PostgreSQL became unhealthy." >&2
|
||||
return 1
|
||||
;;
|
||||
esac
|
||||
sleep 1
|
||||
done
|
||||
wait_for_postgres_final_ready \
|
||||
"$restore_container" \
|
||||
"$evidence_dir/restore-postgres.log" \
|
||||
120 \
|
||||
1
|
||||
|
||||
# The PostGIS image initializes its requested database with PostGIS already
|
||||
# installed. Restore into a database created from template0 instead, so the
|
||||
|
|
|
|||
|
|
@ -128,6 +128,9 @@ docker run --rm \
|
|||
"$SHELLCHECK_IMAGE" \
|
||||
$(find scripts -type f -name '*.sh' -print | sort)
|
||||
|
||||
echo "Checking production restore PostgreSQL readiness"
|
||||
bash test/scripts/production_offsite_backup_readiness_test.sh
|
||||
|
||||
echo "Checking the production read-only load safety boundary"
|
||||
./scripts/production-readonly-load-drill.sh
|
||||
|
||||
|
|
|
|||
|
|
@ -74,6 +74,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
|||
textwrap.dedent(
|
||||
"""\
|
||||
#!/usr/bin/env python3
|
||||
import os
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
|
|
@ -83,6 +84,9 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
|||
raise SystemExit(0)
|
||||
|
||||
command = sys.argv[-1]
|
||||
if "loginctl show-user" in command:
|
||||
print(f"monitor-user:{os.environ.get('FAKE_MONITOR_LINGER', 'yes')}")
|
||||
raise SystemExit(0)
|
||||
if command.startswith("python3 - "):
|
||||
result = subprocess.run(
|
||||
shlex.split(command),
|
||||
|
|
@ -116,7 +120,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
|||
),
|
||||
)
|
||||
|
||||
def run_installer(self, *, override, backup_max_age=None):
|
||||
def run_installer(self, *, override, backup_max_age=None, linger="yes"):
|
||||
self.capture.unlink(missing_ok=True)
|
||||
env = os.environ.copy()
|
||||
env.update(
|
||||
|
|
@ -127,6 +131,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
|||
"PRODUCTION_ENV_PATH": str(self.production_env),
|
||||
"FAKE_MONITOR_CAPTURE": str(self.capture),
|
||||
"MONITOR_INSTALL_WORK_ROOT": str(self.work_root),
|
||||
"FAKE_MONITOR_LINGER": linger,
|
||||
}
|
||||
)
|
||||
if override:
|
||||
|
|
@ -190,6 +195,23 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
|||
self.assertFalse(self.capture.exists())
|
||||
self.assertIn("must be a positive integer", result.stderr)
|
||||
|
||||
def test_linger_disabled_fails_before_external_changes(self):
|
||||
result = self.run_installer(override=True, linger="no")
|
||||
|
||||
self.assertEqual(result.returncode, 2)
|
||||
self.assertFalse(self.capture.exists())
|
||||
self.assertIn("requires systemd user lingering", result.stderr)
|
||||
self.assertIn("sudo loginctl enable-linger 'monitor-user'", result.stderr)
|
||||
|
||||
def test_success_reports_verified_persistent_user_manager(self):
|
||||
result = self.run_installer(override=True, linger="yes")
|
||||
|
||||
self.assertEqual(result.returncode, 0, result.stderr)
|
||||
self.assertIn(
|
||||
"Persistent systemd user manager verified for monitor-user (linger=yes).",
|
||||
result.stdout,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
|
|||
79
test/scripts/production_offsite_backup_readiness_test.sh
Normal file
79
test/scripts/production_offsite_backup_readiness_test.sh
Normal file
|
|
@ -0,0 +1,79 @@
|
|||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/../.." && pwd)
|
||||
|
||||
# ROOT is resolved from this test at runtime.
|
||||
# shellcheck disable=SC1091
|
||||
source "$ROOT/scripts/lib/postgres-readiness.sh"
|
||||
|
||||
mode=eventual_ready
|
||||
sql_attempts=0
|
||||
|
||||
docker() {
|
||||
case "$1" in
|
||||
logs)
|
||||
printf '%s\n' 'PostgreSQL init process complete; ready for start up.'
|
||||
;;
|
||||
inspect)
|
||||
case "$3" in
|
||||
*State.Running*) printf '%s\n' true ;;
|
||||
*State.Health*)
|
||||
if [[ "$mode" == unhealthy ]]; then
|
||||
printf '%s\n' unhealthy
|
||||
else
|
||||
printf '%s\n' healthy
|
||||
fi
|
||||
;;
|
||||
*) return 2 ;;
|
||||
esac
|
||||
;;
|
||||
exec)
|
||||
sql_attempts=$((sql_attempts + 1))
|
||||
if [[ "$mode" == unhealthy ]] || ((sql_attempts == 1)); then
|
||||
return 1
|
||||
fi
|
||||
printf '%s\n' 1
|
||||
;;
|
||||
*)
|
||||
return 2
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
sleep() {
|
||||
:
|
||||
}
|
||||
|
||||
evidence=$(mktemp)
|
||||
trap 'rm -f "$evidence"' EXIT
|
||||
|
||||
wait_for_postgres_final_ready restore-probe "$evidence" 3 0
|
||||
|
||||
if ((sql_attempts != 2)); then
|
||||
printf 'expected two direct SQL readiness attempts, observed %s\n' "$sql_attempts" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ -s "$evidence" ]]; then
|
||||
echo 'success path unexpectedly wrote failure evidence' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mode=unhealthy
|
||||
sql_attempts=0
|
||||
: >"$evidence"
|
||||
|
||||
if wait_for_postgres_final_ready restore-probe "$evidence" 3 0; then
|
||||
echo 'unhealthy final PostgreSQL unexpectedly passed readiness' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ((sql_attempts != 1)); then
|
||||
printf 'expected one SQL attempt before unhealthy failure, observed %s\n' "$sql_attempts" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
grep -Fq 'PostgreSQL init process complete; ready for start up.' "$evidence"
|
||||
|
||||
echo 'Production restore readiness regression passed.'
|
||||
Loading…
Reference in New Issue
Block a user