diff --git a/docs/operations.md b/docs/operations.md index fd017a5..5cd4563 100644 --- a/docs/operations.md +++ b/docs/operations.md @@ -680,6 +680,12 @@ volume on success, failure, or interrupt. Non-secret catalogs, checksums, and restore observations remain under ignored `output/production-operations//`. +After all of those checks pass, `run` atomically publishes a mode-`0600` +heartbeat on the independent repository host. The heartbeat contains only the +Restic snapshot identifier, verification time, and `restore_verified: true`; +it contains no database rows, account identifiers, recipient addresses, or +credentials. A failed backup or failed restore never advances the heartbeat. + Install the daily user-systemd timer only after one manual `run` succeeds: ```bash @@ -712,6 +718,21 @@ configuration. Readiness or metrics-scrape failure changes the monitor state to down. Email is sent once when that state changes to down and once when it recovers; repeated down checks do not send repeated alerts. +The monitor can also treat a missing, invalid, unverified, or stale backup +heartbeat as down. This is intentionally not enabled with an invented default: +the maximum acceptable age is the operator's RPO/alerting decision. After that +decision is recorded, reinstall the monitor with the chosen positive number of +seconds: + +```bash +MONITOR_BACKUP_MAX_AGE_SECONDS= \ + ./scripts/install-production-external-monitor.sh +``` + +The installed configuration then records the exact threshold and heartbeat +path. A stale backup produces one state-transition alert; a later successful, +restore-verified backup produces one recovery notification. + For each observed application node, the monitor keeps a baseline of aggregate HTTP exception, failed Oban job, and failed/exceptional email-delivery counters. It sends one aggregate notification when one of those counters increases. A diff --git a/scripts/init-production-operations.sh b/scripts/init-production-operations.sh index 968f857..b64d0f8 100755 --- a/scripts/init-production-operations.sh +++ b/scripts/init-production-operations.sh @@ -56,6 +56,7 @@ OFFSITE_RESTIC_REPOSITORY=sftp:$repository_target:backups/who_need_help-producti OFFSITE_RESTIC_PASSWORD_FILE=$password_file OFFSITE_RESTIC_HOST=who-need-help-production OFFSITE_RESTIC_TAG=who-need-help-production +OFFSITE_BACKUP_HEARTBEAT_PATH=/home/simple/.local/state/who-need-help/production-backup.json BACKUP_ON_CALENDAR=daily EOF chmod 600 "$config" diff --git a/scripts/install-production-external-monitor.sh b/scripts/install-production-external-monitor.sh index e668627..3054197 100755 --- a/scripts/install-production-external-monitor.sh +++ b/scripts/install-production-external-monitor.sh @@ -16,6 +16,8 @@ metrics_url=${MONITOR_METRICS_URL:-https://whoneedhelp.com/metrics} monitor_calendar=${MONITOR_ON_CALENDAR:-'*:0/1'} health_timeout=${MONITOR_HEALTH_TIMEOUT_SECONDS:-3} metrics_timeout=${MONITOR_METRICS_TIMEOUT_SECONDS:-3} +backup_heartbeat_path=${MONITOR_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json} +backup_max_age=${MONITOR_BACKUP_MAX_AGE_SECONDS:-} for command in awk install mktemp python3 realpath scp ssh stat; do command -v "$command" >/dev/null 2>&1 || { @@ -75,6 +77,17 @@ case "$metrics_timeout" in 0) echo "MONITOR_METRICS_TIMEOUT_SECONDS must be a positive integer." >&2; exit 2 ;; esac +if [ -n "$backup_max_age" ]; then + case "$backup_max_age" in + *[!0-9]*) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;; + 0) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;; + esac + if [ "$backup_heartbeat_path" != /home/simple/.local/state/who-need-help/production-backup.json ]; then + echo "MONITOR_BACKUP_HEARTBEAT_PATH is outside the reviewed monitor state path." >&2 + exit 2 + fi +fi + work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX") config="$work_dir/monitor.json" service="$work_dir/who-need-help-production-monitor.service" @@ -87,12 +100,21 @@ cleanup() { trap cleanup 0 HUP INT TERM ssh -o BatchMode=yes "$source_target" \ - "python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source'" >"$config" <<'PY' + "python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source' '$backup_heartbeat_path' '$backup_max_age'" >"$config" <<'PY' import json import shlex import sys -path, health_url, metrics_url, health_timeout, metrics_timeout, smtp_source = sys.argv[1:] +( + path, + health_url, + metrics_url, + health_timeout, + metrics_timeout, + smtp_source, + backup_heartbeat_path, + backup_max_age, +) = sys.argv[1:] smtp_keys = { "SMTP_RELAY", "SMTP_PORT", @@ -148,6 +170,11 @@ payload = { "metrics_url": metrics_url, "smtp": smtp, } +if backup_max_age: + payload["backup"] = { + "heartbeat_path": backup_heartbeat_path, + "max_age_seconds": int(backup_max_age), + } json.dump(payload, sys.stdout, ensure_ascii=False, indent=2, sort_keys=True) sys.stdout.write("\n") PY @@ -198,6 +225,12 @@ printf 'Health URL: %s\n' "$monitor_url" printf 'Metrics URL: %s\n' "$metrics_url" printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \ "$monitor_calendar" "$health_timeout" "$metrics_timeout" +if [ -n "$backup_max_age" ]; then + printf 'Backup freshness: heartbeat %s; operator-selected maximum age %ss.\n' \ + "$backup_heartbeat_path" "$backup_max_age" +else + echo "Backup freshness monitoring is not enabled because no operator-selected maximum age was supplied." +fi echo "The SMTP and metrics credentials are stored only in a mode-0600 configuration on the external monitor host." if [ -n "$monitor_smtp_values_file" ]; then echo "The external monitor uses the independently supplied SMTP credential set." diff --git a/scripts/production-external-monitor.py b/scripts/production-external-monitor.py index f7a20f6..6cac61f 100755 --- a/scripts/production-external-monitor.py +++ b/scripts/production-external-monitor.py @@ -212,6 +212,64 @@ def check_metrics(config: dict[str, Any]) -> tuple[str, str, str, dict[str, floa return "down", f"{type(error).__name__}: {str(error)[:500]}", "unknown", {} +def parse_utc_timestamp(value: Any) -> datetime: + if not isinstance(value, str) or not value.strip(): + raise ValueError("Missing verified backup timestamp") + + parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00")) + if parsed.tzinfo is None: + raise ValueError("Backup timestamp must include a UTC offset") + return parsed.astimezone(timezone.utc) + + +def check_backup_freshness( + config: dict[str, Any], *, now: datetime | None = None +) -> tuple[str, str]: + backup = config.get("backup") + if backup is None: + return "disabled", "backup freshness monitoring is not configured" + if not isinstance(backup, dict): + raise ValueError("Backup monitoring configuration must be an object") + + heartbeat_path = Path(required_text(backup, "heartbeat_path")) + if not heartbeat_path.is_absolute(): + raise ValueError("Backup heartbeat path must be absolute") + + raw_max_age = backup.get("max_age_seconds") + if isinstance(raw_max_age, bool): + raise ValueError("Backup max_age_seconds must be a positive integer") + try: + max_age_seconds = int(raw_max_age) + except (TypeError, ValueError) as error: + raise ValueError("Backup max_age_seconds must be a positive integer") from error + if max_age_seconds <= 0: + raise ValueError("Backup max_age_seconds must be positive") + + try: + heartbeat = read_json(heartbeat_path) + if heartbeat.get("restore_verified") is not True: + return "down", "latest backup heartbeat is not restore-verified" + required_text(heartbeat, "snapshot_id") + verified_at = parse_utc_timestamp(heartbeat.get("verified_at")) + except (OSError, ValueError, json.JSONDecodeError) as error: + return "down", f"{type(error).__name__}: {str(error)[:500]}" + + observed_at = (now or datetime.now(timezone.utc)).astimezone(timezone.utc) + if verified_at > observed_at: + return "down", "latest backup heartbeat timestamp is in the future" + + age_seconds = max(0, int((observed_at - verified_at).total_seconds())) + if age_seconds > max_age_seconds: + return ( + "down", + f"latest restore-verified backup is stale (age={age_seconds}s; max={max_age_seconds}s)", + ) + return ( + "up", + f"latest restore-verified backup is fresh (age={age_seconds}s; max={max_age_seconds}s)", + ) + + def monitor(config_path: Path, state_path: Path) -> int: config = read_json(config_path) previous: dict[str, Any] = {} @@ -220,8 +278,21 @@ def monitor(config_path: Path, state_path: Path) -> int: health_status, health_detail = check_health(config) metrics_status, metrics_detail, metrics_node, counters = check_metrics(config) - status = "up" if health_status == "up" and metrics_status == "up" else "down" - detail = f"readiness={health_status} ({health_detail}); metrics={metrics_status} ({metrics_detail})" + backup_status, backup_detail = check_backup_freshness(config) + status = ( + "up" + if health_status == "up" + and metrics_status == "up" + and backup_status in {"up", "disabled"} + else "down" + ) + detail = "; ".join( + [ + f"readiness={health_status} ({health_detail})", + f"metrics={metrics_status} ({metrics_detail})", + f"backup={backup_status} ({backup_detail})", + ] + ) previous_status = previous.get("status") if status != previous_status: @@ -231,7 +302,7 @@ def monitor(config_path: Path, state_path: Path) -> int: "[Who Need Help] Production monitor is DOWN", "\n".join( [ - "The independent production readiness or metrics check failed.", + "The independent production readiness, metrics, or backup-freshness check failed.", f"URL: {required_text(config, 'health_url')}", f"Metrics URL: {required_text(config, 'metrics_url')}", f"Observed at: {utc_now()}", @@ -246,7 +317,7 @@ def monitor(config_path: Path, state_path: Path) -> int: "[Who Need Help] Production monitor recovered", "\n".join( [ - "The independent production readiness and metrics checks recovered.", + "The independent production readiness, metrics, and configured backup-freshness checks recovered.", f"URL: {required_text(config, 'health_url')}", f"Metrics URL: {required_text(config, 'metrics_url')}", f"Observed at: {utc_now()}", @@ -286,6 +357,7 @@ def monitor(config_path: Path, state_path: Path) -> int: state_path, { "checked_at": utc_now(), + "backup_status": backup_status, "detail": detail, "health_status": health_status, "metrics_by_node": metrics_by_node, diff --git a/scripts/production-offsite-backup.sh b/scripts/production-offsite-backup.sh index 894a849..0ea3a2f 100755 --- a/scripts/production-offsite-backup.sh +++ b/scripts/production-offsite-backup.sh @@ -40,6 +40,7 @@ source "$config" : "${OFFSITE_RESTIC_PASSWORD_FILE:?Set OFFSITE_RESTIC_PASSWORD_FILE}" : "${OFFSITE_RESTIC_HOST:?Set OFFSITE_RESTIC_HOST}" : "${OFFSITE_RESTIC_TAG:?Set OFFSITE_RESTIC_TAG}" +OFFSITE_BACKUP_HEARTBEAT_PATH=${OFFSITE_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json} if [[ "$PRODUCTION_EXPECTED_ENVIRONMENT" != production ]]; then echo "The source environment must be exactly production." >&2 @@ -62,6 +63,11 @@ case "$OFFSITE_RESTIC_REPOSITORY" in ;; esac +if [[ "$OFFSITE_BACKUP_HEARTBEAT_PATH" != /home/simple/.local/state/who-need-help/production-backup.json ]]; then + echo "The backup heartbeat path is outside the reviewed independent-monitor state path." >&2 + exit 2 +fi + if [[ ! -f "$OFFSITE_RESTIC_PASSWORD_FILE" ]] || [[ "$(stat -c '%a' "$OFFSITE_RESTIC_PASSWORD_FILE")" != 600 ]]; then echo "The Restic password file must exist with mode 0600." >&2 @@ -182,6 +188,8 @@ restore_container="wnh-production-restore-$run_id" restore_volume="wnh_production_restore_${run_id//[^a-zA-Z0-9]/_}" restore_started=false restore_volume_created=false +heartbeat_staged=false +heartbeat_remote_tmp="$OFFSITE_BACKUP_HEARTBEAT_PATH.tmp-$run_id" mkdir -p "$evidence_dir" chmod 700 "$ROOT/output" "$ROOT/output/production-operations" "$evidence_dir" "$work_dir" @@ -199,6 +207,10 @@ cleanup() { "rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" \ >/dev/null 2>&1 || true fi + if [[ "$heartbeat_staged" == true ]]; then + ssh -o BatchMode=yes "$repository_alias" \ + "rm -f -- '$heartbeat_remote_tmp'" >/dev/null 2>&1 || true + fi rm -rf "$work_dir" } trap cleanup EXIT HUP INT TERM @@ -379,6 +391,21 @@ ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \ "rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" remote_created=false +heartbeat="$work_dir/production-backup-heartbeat.json" +jq -n \ + --arg snapshot_id "$snapshot_id" \ + --arg verified_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \ + '{snapshot_id: $snapshot_id, verified_at: $verified_at, restore_verified: true}' \ + >"$heartbeat" +chmod 600 "$heartbeat" +ssh -o BatchMode=yes "$repository_alias" \ + "install -d -m 700 '$(dirname -- "$OFFSITE_BACKUP_HEARTBEAT_PATH")'" +heartbeat_staged=true +scp -q "$heartbeat" "$repository_alias:$heartbeat_remote_tmp" +ssh -o BatchMode=yes "$repository_alias" \ + "chmod 600 '$heartbeat_remote_tmp' && mv -f -- '$heartbeat_remote_tmp' '$OFFSITE_BACKUP_HEARTBEAT_PATH'" +heartbeat_staged=false + echo "Encrypted off-server backup and isolated restore drill passed." printf 'Snapshot: %s\n' "$snapshot_id" printf 'Non-secret evidence: %s\n' "$evidence_dir" diff --git a/test/scripts/install_production_external_monitor_test.py b/test/scripts/install_production_external_monitor_test.py index 61e9e1f..48522a9 100644 --- a/test/scripts/install_production_external_monitor_test.py +++ b/test/scripts/install_production_external_monitor_test.py @@ -116,7 +116,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase): ), ) - def run_installer(self, *, override): + def run_installer(self, *, override, backup_max_age=None): self.capture.unlink(missing_ok=True) env = os.environ.copy() env.update( @@ -133,6 +133,10 @@ class InstallProductionExternalMonitorTest(unittest.TestCase): env["MONITOR_SMTP_VALUES_FILE"] = str(self.monitor_values) else: env.pop("MONITOR_SMTP_VALUES_FILE", None) + if backup_max_age is not None: + env["MONITOR_BACKUP_MAX_AGE_SECONDS"] = str(backup_max_age) + else: + env.pop("MONITOR_BACKUP_MAX_AGE_SECONDS", None) return subprocess.run( [str(INSTALLER)], cwd=ROOT, @@ -165,6 +169,27 @@ class InstallProductionExternalMonitorTest(unittest.TestCase): self.assertNotIn("application-secret", result.stdout + result.stderr) self.assertIn("mirrors the production application SMTP", result.stdout) + def test_operator_selected_backup_age_reaches_monitor_config(self): + result = self.run_installer(override=True, backup_max_age=172800) + + self.assertEqual(result.returncode, 0, result.stderr) + captured = json.loads(self.capture.read_text(encoding="utf-8")) + self.assertEqual( + captured["backup"], + { + "heartbeat_path": "/home/simple/.local/state/who-need-help/production-backup.json", + "max_age_seconds": 172800, + }, + ) + self.assertIn("operator-selected maximum age 172800s", result.stdout) + + def test_invalid_backup_age_fails_before_external_changes(self): + result = self.run_installer(override=True, backup_max_age="tomorrow") + + self.assertEqual(result.returncode, 2) + self.assertFalse(self.capture.exists()) + self.assertIn("must be a positive integer", result.stderr) + if __name__ == "__main__": unittest.main() diff --git a/test/scripts/production_external_monitor_test.py b/test/scripts/production_external_monitor_test.py index 0adb3d2..6de82a7 100644 --- a/test/scripts/production_external_monitor_test.py +++ b/test/scripts/production_external_monitor_test.py @@ -4,6 +4,7 @@ import importlib.util import json import tempfile import unittest +from datetime import datetime, timezone from pathlib import Path from unittest import mock @@ -16,6 +17,95 @@ SPEC.loader.exec_module(MONITOR) class ProductionExternalMonitorTest(unittest.TestCase): + def test_backup_freshness_is_disabled_without_an_operator_policy(self): + self.assertEqual( + MONITOR.check_backup_freshness({}), + ("disabled", "backup freshness monitoring is not configured"), + ) + + def test_backup_freshness_rejects_missing_or_boolean_age(self): + for value in (None, True): + with self.subTest(value=value): + with self.assertRaisesRegex(ValueError, "must be a positive integer"): + MONITOR.check_backup_freshness( + { + "backup": { + "heartbeat_path": "/tmp/backup.json", + "max_age_seconds": value, + } + } + ) + + def test_backup_freshness_accepts_recent_restore_verified_heartbeat(self): + with tempfile.TemporaryDirectory() as directory: + heartbeat = Path(directory) / "backup.json" + heartbeat.write_text( + json.dumps( + { + "restore_verified": True, + "snapshot_id": "abc123", + "verified_at": "2026-08-12T00:00:00Z", + } + ), + encoding="utf-8", + ) + status, detail = MONITOR.check_backup_freshness( + { + "backup": { + "heartbeat_path": str(heartbeat), + "max_age_seconds": 86_400, + } + }, + now=datetime(2026, 8, 12, 12, 0, tzinfo=timezone.utc), + ) + + self.assertEqual(status, "up") + self.assertIn("age=43200s", detail) + + def test_backup_freshness_rejects_stale_or_unverified_heartbeat(self): + with tempfile.TemporaryDirectory() as directory: + heartbeat = Path(directory) / "backup.json" + config = { + "backup": { + "heartbeat_path": str(heartbeat), + "max_age_seconds": 3_600, + } + } + heartbeat.write_text( + json.dumps( + { + "restore_verified": True, + "snapshot_id": "abc123", + "verified_at": "2026-08-12T00:00:00Z", + } + ), + encoding="utf-8", + ) + status, detail = MONITOR.check_backup_freshness( + config, + now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc), + ) + self.assertEqual(status, "down") + self.assertIn("stale", detail) + + heartbeat.write_text( + json.dumps( + { + "restore_verified": False, + "snapshot_id": "abc123", + "verified_at": "2026-08-12T02:00:00Z", + } + ), + encoding="utf-8", + ) + self.assertEqual( + MONITOR.check_backup_freshness( + config, + now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc), + ), + ("down", "latest backup heartbeat is not restore-verified"), + ) + def test_parses_only_aggregate_failure_counters(self): payload = """ # TYPE who_need_help_http_exceptions_total counter @@ -80,6 +170,11 @@ who_need_help_http_requests_total 999 "check_metrics", return_value=("up", "parsed", "node-a", {"failure": 2.0}), ), + mock.patch.object( + MONITOR, + "check_backup_freshness", + return_value=("disabled", "not configured"), + ), mock.patch.object( MONITOR, "send_message", @@ -98,6 +193,11 @@ who_need_help_http_requests_total 999 "check_metrics", return_value=("up", "parsed", "node-a", {"failure": 5.0}), ), + mock.patch.object( + MONITOR, + "check_backup_freshness", + return_value=("disabled", "not configured"), + ), mock.patch.object( MONITOR, "send_message", @@ -133,6 +233,11 @@ who_need_help_http_requests_total 999 "check_metrics", return_value=("down", "HTTP 401", "unknown", {}), ), + mock.patch.object( + MONITOR, + "check_backup_freshness", + return_value=("disabled", "not configured"), + ), mock.patch.object( MONITOR, "send_message", @@ -153,6 +258,75 @@ who_need_help_http_requests_total 999 "check_metrics", return_value=("up", "parsed", "node-a", {}), ), + mock.patch.object( + MONITOR, + "check_backup_freshness", + return_value=("disabled", "not configured"), + ), + mock.patch.object( + MONITOR, + "send_message", + side_effect=lambda _config, subject, body: messages.append((subject, body)), + ), + mock.patch("builtins.print"), + ): + self.assertEqual(MONITOR.monitor(config_path, state_path), 0) + + self.assertEqual(len(messages), 2) + self.assertIn("recovered", messages[1][0]) + + def test_monitor_alerts_for_stale_backup_then_recovers(self): + with tempfile.TemporaryDirectory() as directory: + config_path = Path(directory) / "config.json" + state_path = Path(directory) / "state.json" + config_path.write_text( + json.dumps( + { + "health_url": "https://example.test/healthz/ready", + "metrics_url": "https://example.test/metrics", + } + ), + encoding="utf-8", + ) + messages = [] + + with ( + mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")), + mock.patch.object( + MONITOR, + "check_metrics", + return_value=("up", "parsed", "node-a", {}), + ), + mock.patch.object( + MONITOR, + "check_backup_freshness", + return_value=("down", "latest backup is stale"), + ), + mock.patch.object( + MONITOR, + "send_message", + side_effect=lambda _config, subject, body: messages.append((subject, body)), + ), + mock.patch("builtins.print"), + ): + self.assertEqual(MONITOR.monitor(config_path, state_path), 1) + + self.assertEqual(len(messages), 1) + self.assertIn("DOWN", messages[0][0]) + self.assertIn("backup=down", messages[0][1]) + + with ( + mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")), + mock.patch.object( + MONITOR, + "check_metrics", + return_value=("up", "parsed", "node-a", {}), + ), + mock.patch.object( + MONITOR, + "check_backup_freshness", + return_value=("up", "latest backup is fresh"), + ), mock.patch.object( MONITOR, "send_message",