Monitor restore-verified backup freshness

This commit is contained in:
SimpleTest 2026-08-12 18:18:29 +03:00
parent 0a575a6b33
commit 346b03695f
7 changed files with 360 additions and 7 deletions

View File

@ -680,6 +680,12 @@ volume on success, failure, or interrupt. Non-secret catalogs, checksums, and
restore observations remain under ignored
`output/production-operations/<run-id>/`.
After all of those checks pass, `run` atomically publishes a mode-`0600`
heartbeat on the independent repository host. The heartbeat contains only the
Restic snapshot identifier, verification time, and `restore_verified: true`;
it contains no database rows, account identifiers, recipient addresses, or
credentials. A failed backup or failed restore never advances the heartbeat.
Install the daily user-systemd timer only after one manual `run` succeeds:
```bash
@ -712,6 +718,21 @@ configuration. Readiness or metrics-scrape failure changes the monitor state to
down. Email is sent once when that state changes to down and once when it
recovers; repeated down checks do not send repeated alerts.
The monitor can also treat a missing, invalid, unverified, or stale backup
heartbeat as down. This is intentionally not enabled with an invented default:
the maximum acceptable age is the operator's RPO/alerting decision. After that
decision is recorded, reinstall the monitor with the chosen positive number of
seconds:
```bash
MONITOR_BACKUP_MAX_AGE_SECONDS=<operator-selected-seconds> \
./scripts/install-production-external-monitor.sh
```
The installed configuration then records the exact threshold and heartbeat
path. A stale backup produces one state-transition alert; a later successful,
restore-verified backup produces one recovery notification.
For each observed application node, the monitor keeps a baseline of aggregate
HTTP exception, failed Oban job, and failed/exceptional email-delivery counters.
It sends one aggregate notification when one of those counters increases. A

View File

@ -56,6 +56,7 @@ OFFSITE_RESTIC_REPOSITORY=sftp:$repository_target:backups/who_need_help-producti
OFFSITE_RESTIC_PASSWORD_FILE=$password_file
OFFSITE_RESTIC_HOST=who-need-help-production
OFFSITE_RESTIC_TAG=who-need-help-production
OFFSITE_BACKUP_HEARTBEAT_PATH=/home/simple/.local/state/who-need-help/production-backup.json
BACKUP_ON_CALENDAR=daily
EOF
chmod 600 "$config"

View File

@ -16,6 +16,8 @@ metrics_url=${MONITOR_METRICS_URL:-https://whoneedhelp.com/metrics}
monitor_calendar=${MONITOR_ON_CALENDAR:-'*:0/1'}
health_timeout=${MONITOR_HEALTH_TIMEOUT_SECONDS:-3}
metrics_timeout=${MONITOR_METRICS_TIMEOUT_SECONDS:-3}
backup_heartbeat_path=${MONITOR_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json}
backup_max_age=${MONITOR_BACKUP_MAX_AGE_SECONDS:-}
for command in awk install mktemp python3 realpath scp ssh stat; do
command -v "$command" >/dev/null 2>&1 || {
@ -75,6 +77,17 @@ case "$metrics_timeout" in
0) echo "MONITOR_METRICS_TIMEOUT_SECONDS must be a positive integer." >&2; exit 2 ;;
esac
if [ -n "$backup_max_age" ]; then
case "$backup_max_age" in
*[!0-9]*) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;;
0) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;;
esac
if [ "$backup_heartbeat_path" != /home/simple/.local/state/who-need-help/production-backup.json ]; then
echo "MONITOR_BACKUP_HEARTBEAT_PATH is outside the reviewed monitor state path." >&2
exit 2
fi
fi
work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX")
config="$work_dir/monitor.json"
service="$work_dir/who-need-help-production-monitor.service"
@ -87,12 +100,21 @@ cleanup() {
trap cleanup 0 HUP INT TERM
ssh -o BatchMode=yes "$source_target" \
"python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source'" >"$config" <<'PY'
"python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source' '$backup_heartbeat_path' '$backup_max_age'" >"$config" <<'PY'
import json
import shlex
import sys
path, health_url, metrics_url, health_timeout, metrics_timeout, smtp_source = sys.argv[1:]
(
path,
health_url,
metrics_url,
health_timeout,
metrics_timeout,
smtp_source,
backup_heartbeat_path,
backup_max_age,
) = sys.argv[1:]
smtp_keys = {
"SMTP_RELAY",
"SMTP_PORT",
@ -148,6 +170,11 @@ payload = {
"metrics_url": metrics_url,
"smtp": smtp,
}
if backup_max_age:
payload["backup"] = {
"heartbeat_path": backup_heartbeat_path,
"max_age_seconds": int(backup_max_age),
}
json.dump(payload, sys.stdout, ensure_ascii=False, indent=2, sort_keys=True)
sys.stdout.write("\n")
PY
@ -198,6 +225,12 @@ printf 'Health URL: %s\n' "$monitor_url"
printf 'Metrics URL: %s\n' "$metrics_url"
printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \
"$monitor_calendar" "$health_timeout" "$metrics_timeout"
if [ -n "$backup_max_age" ]; then
printf 'Backup freshness: heartbeat %s; operator-selected maximum age %ss.\n' \
"$backup_heartbeat_path" "$backup_max_age"
else
echo "Backup freshness monitoring is not enabled because no operator-selected maximum age was supplied."
fi
echo "The SMTP and metrics credentials are stored only in a mode-0600 configuration on the external monitor host."
if [ -n "$monitor_smtp_values_file" ]; then
echo "The external monitor uses the independently supplied SMTP credential set."

View File

@ -212,6 +212,64 @@ def check_metrics(config: dict[str, Any]) -> tuple[str, str, str, dict[str, floa
return "down", f"{type(error).__name__}: {str(error)[:500]}", "unknown", {}
def parse_utc_timestamp(value: Any) -> datetime:
if not isinstance(value, str) or not value.strip():
raise ValueError("Missing verified backup timestamp")
parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00"))
if parsed.tzinfo is None:
raise ValueError("Backup timestamp must include a UTC offset")
return parsed.astimezone(timezone.utc)
def check_backup_freshness(
config: dict[str, Any], *, now: datetime | None = None
) -> tuple[str, str]:
backup = config.get("backup")
if backup is None:
return "disabled", "backup freshness monitoring is not configured"
if not isinstance(backup, dict):
raise ValueError("Backup monitoring configuration must be an object")
heartbeat_path = Path(required_text(backup, "heartbeat_path"))
if not heartbeat_path.is_absolute():
raise ValueError("Backup heartbeat path must be absolute")
raw_max_age = backup.get("max_age_seconds")
if isinstance(raw_max_age, bool):
raise ValueError("Backup max_age_seconds must be a positive integer")
try:
max_age_seconds = int(raw_max_age)
except (TypeError, ValueError) as error:
raise ValueError("Backup max_age_seconds must be a positive integer") from error
if max_age_seconds <= 0:
raise ValueError("Backup max_age_seconds must be positive")
try:
heartbeat = read_json(heartbeat_path)
if heartbeat.get("restore_verified") is not True:
return "down", "latest backup heartbeat is not restore-verified"
required_text(heartbeat, "snapshot_id")
verified_at = parse_utc_timestamp(heartbeat.get("verified_at"))
except (OSError, ValueError, json.JSONDecodeError) as error:
return "down", f"{type(error).__name__}: {str(error)[:500]}"
observed_at = (now or datetime.now(timezone.utc)).astimezone(timezone.utc)
if verified_at > observed_at:
return "down", "latest backup heartbeat timestamp is in the future"
age_seconds = max(0, int((observed_at - verified_at).total_seconds()))
if age_seconds > max_age_seconds:
return (
"down",
f"latest restore-verified backup is stale (age={age_seconds}s; max={max_age_seconds}s)",
)
return (
"up",
f"latest restore-verified backup is fresh (age={age_seconds}s; max={max_age_seconds}s)",
)
def monitor(config_path: Path, state_path: Path) -> int:
config = read_json(config_path)
previous: dict[str, Any] = {}
@ -220,8 +278,21 @@ def monitor(config_path: Path, state_path: Path) -> int:
health_status, health_detail = check_health(config)
metrics_status, metrics_detail, metrics_node, counters = check_metrics(config)
status = "up" if health_status == "up" and metrics_status == "up" else "down"
detail = f"readiness={health_status} ({health_detail}); metrics={metrics_status} ({metrics_detail})"
backup_status, backup_detail = check_backup_freshness(config)
status = (
"up"
if health_status == "up"
and metrics_status == "up"
and backup_status in {"up", "disabled"}
else "down"
)
detail = "; ".join(
[
f"readiness={health_status} ({health_detail})",
f"metrics={metrics_status} ({metrics_detail})",
f"backup={backup_status} ({backup_detail})",
]
)
previous_status = previous.get("status")
if status != previous_status:
@ -231,7 +302,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
"[Who Need Help] Production monitor is DOWN",
"\n".join(
[
"The independent production readiness or metrics check failed.",
"The independent production readiness, metrics, or backup-freshness check failed.",
f"URL: {required_text(config, 'health_url')}",
f"Metrics URL: {required_text(config, 'metrics_url')}",
f"Observed at: {utc_now()}",
@ -246,7 +317,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
"[Who Need Help] Production monitor recovered",
"\n".join(
[
"The independent production readiness and metrics checks recovered.",
"The independent production readiness, metrics, and configured backup-freshness checks recovered.",
f"URL: {required_text(config, 'health_url')}",
f"Metrics URL: {required_text(config, 'metrics_url')}",
f"Observed at: {utc_now()}",
@ -286,6 +357,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
state_path,
{
"checked_at": utc_now(),
"backup_status": backup_status,
"detail": detail,
"health_status": health_status,
"metrics_by_node": metrics_by_node,

View File

@ -40,6 +40,7 @@ source "$config"
: "${OFFSITE_RESTIC_PASSWORD_FILE:?Set OFFSITE_RESTIC_PASSWORD_FILE}"
: "${OFFSITE_RESTIC_HOST:?Set OFFSITE_RESTIC_HOST}"
: "${OFFSITE_RESTIC_TAG:?Set OFFSITE_RESTIC_TAG}"
OFFSITE_BACKUP_HEARTBEAT_PATH=${OFFSITE_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json}
if [[ "$PRODUCTION_EXPECTED_ENVIRONMENT" != production ]]; then
echo "The source environment must be exactly production." >&2
@ -62,6 +63,11 @@ case "$OFFSITE_RESTIC_REPOSITORY" in
;;
esac
if [[ "$OFFSITE_BACKUP_HEARTBEAT_PATH" != /home/simple/.local/state/who-need-help/production-backup.json ]]; then
echo "The backup heartbeat path is outside the reviewed independent-monitor state path." >&2
exit 2
fi
if [[ ! -f "$OFFSITE_RESTIC_PASSWORD_FILE" ]] ||
[[ "$(stat -c '%a' "$OFFSITE_RESTIC_PASSWORD_FILE")" != 600 ]]; then
echo "The Restic password file must exist with mode 0600." >&2
@ -182,6 +188,8 @@ restore_container="wnh-production-restore-$run_id"
restore_volume="wnh_production_restore_${run_id//[^a-zA-Z0-9]/_}"
restore_started=false
restore_volume_created=false
heartbeat_staged=false
heartbeat_remote_tmp="$OFFSITE_BACKUP_HEARTBEAT_PATH.tmp-$run_id"
mkdir -p "$evidence_dir"
chmod 700 "$ROOT/output" "$ROOT/output/production-operations" "$evidence_dir" "$work_dir"
@ -199,6 +207,10 @@ cleanup() {
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" \
>/dev/null 2>&1 || true
fi
if [[ "$heartbeat_staged" == true ]]; then
ssh -o BatchMode=yes "$repository_alias" \
"rm -f -- '$heartbeat_remote_tmp'" >/dev/null 2>&1 || true
fi
rm -rf "$work_dir"
}
trap cleanup EXIT HUP INT TERM
@ -379,6 +391,21 @@ ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'"
remote_created=false
heartbeat="$work_dir/production-backup-heartbeat.json"
jq -n \
--arg snapshot_id "$snapshot_id" \
--arg verified_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \
'{snapshot_id: $snapshot_id, verified_at: $verified_at, restore_verified: true}' \
>"$heartbeat"
chmod 600 "$heartbeat"
ssh -o BatchMode=yes "$repository_alias" \
"install -d -m 700 '$(dirname -- "$OFFSITE_BACKUP_HEARTBEAT_PATH")'"
heartbeat_staged=true
scp -q "$heartbeat" "$repository_alias:$heartbeat_remote_tmp"
ssh -o BatchMode=yes "$repository_alias" \
"chmod 600 '$heartbeat_remote_tmp' && mv -f -- '$heartbeat_remote_tmp' '$OFFSITE_BACKUP_HEARTBEAT_PATH'"
heartbeat_staged=false
echo "Encrypted off-server backup and isolated restore drill passed."
printf 'Snapshot: %s\n' "$snapshot_id"
printf 'Non-secret evidence: %s\n' "$evidence_dir"

View File

@ -116,7 +116,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
),
)
def run_installer(self, *, override):
def run_installer(self, *, override, backup_max_age=None):
self.capture.unlink(missing_ok=True)
env = os.environ.copy()
env.update(
@ -133,6 +133,10 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
env["MONITOR_SMTP_VALUES_FILE"] = str(self.monitor_values)
else:
env.pop("MONITOR_SMTP_VALUES_FILE", None)
if backup_max_age is not None:
env["MONITOR_BACKUP_MAX_AGE_SECONDS"] = str(backup_max_age)
else:
env.pop("MONITOR_BACKUP_MAX_AGE_SECONDS", None)
return subprocess.run(
[str(INSTALLER)],
cwd=ROOT,
@ -165,6 +169,27 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
self.assertNotIn("application-secret", result.stdout + result.stderr)
self.assertIn("mirrors the production application SMTP", result.stdout)
def test_operator_selected_backup_age_reaches_monitor_config(self):
result = self.run_installer(override=True, backup_max_age=172800)
self.assertEqual(result.returncode, 0, result.stderr)
captured = json.loads(self.capture.read_text(encoding="utf-8"))
self.assertEqual(
captured["backup"],
{
"heartbeat_path": "/home/simple/.local/state/who-need-help/production-backup.json",
"max_age_seconds": 172800,
},
)
self.assertIn("operator-selected maximum age 172800s", result.stdout)
def test_invalid_backup_age_fails_before_external_changes(self):
result = self.run_installer(override=True, backup_max_age="tomorrow")
self.assertEqual(result.returncode, 2)
self.assertFalse(self.capture.exists())
self.assertIn("must be a positive integer", result.stderr)
if __name__ == "__main__":
unittest.main()

View File

@ -4,6 +4,7 @@ import importlib.util
import json
import tempfile
import unittest
from datetime import datetime, timezone
from pathlib import Path
from unittest import mock
@ -16,6 +17,95 @@ SPEC.loader.exec_module(MONITOR)
class ProductionExternalMonitorTest(unittest.TestCase):
def test_backup_freshness_is_disabled_without_an_operator_policy(self):
self.assertEqual(
MONITOR.check_backup_freshness({}),
("disabled", "backup freshness monitoring is not configured"),
)
def test_backup_freshness_rejects_missing_or_boolean_age(self):
for value in (None, True):
with self.subTest(value=value):
with self.assertRaisesRegex(ValueError, "must be a positive integer"):
MONITOR.check_backup_freshness(
{
"backup": {
"heartbeat_path": "/tmp/backup.json",
"max_age_seconds": value,
}
}
)
def test_backup_freshness_accepts_recent_restore_verified_heartbeat(self):
with tempfile.TemporaryDirectory() as directory:
heartbeat = Path(directory) / "backup.json"
heartbeat.write_text(
json.dumps(
{
"restore_verified": True,
"snapshot_id": "abc123",
"verified_at": "2026-08-12T00:00:00Z",
}
),
encoding="utf-8",
)
status, detail = MONITOR.check_backup_freshness(
{
"backup": {
"heartbeat_path": str(heartbeat),
"max_age_seconds": 86_400,
}
},
now=datetime(2026, 8, 12, 12, 0, tzinfo=timezone.utc),
)
self.assertEqual(status, "up")
self.assertIn("age=43200s", detail)
def test_backup_freshness_rejects_stale_or_unverified_heartbeat(self):
with tempfile.TemporaryDirectory() as directory:
heartbeat = Path(directory) / "backup.json"
config = {
"backup": {
"heartbeat_path": str(heartbeat),
"max_age_seconds": 3_600,
}
}
heartbeat.write_text(
json.dumps(
{
"restore_verified": True,
"snapshot_id": "abc123",
"verified_at": "2026-08-12T00:00:00Z",
}
),
encoding="utf-8",
)
status, detail = MONITOR.check_backup_freshness(
config,
now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc),
)
self.assertEqual(status, "down")
self.assertIn("stale", detail)
heartbeat.write_text(
json.dumps(
{
"restore_verified": False,
"snapshot_id": "abc123",
"verified_at": "2026-08-12T02:00:00Z",
}
),
encoding="utf-8",
)
self.assertEqual(
MONITOR.check_backup_freshness(
config,
now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc),
),
("down", "latest backup heartbeat is not restore-verified"),
)
def test_parses_only_aggregate_failure_counters(self):
payload = """
# TYPE who_need_help_http_exceptions_total counter
@ -80,6 +170,11 @@ who_need_help_http_requests_total 999
"check_metrics",
return_value=("up", "parsed", "node-a", {"failure": 2.0}),
),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("disabled", "not configured"),
),
mock.patch.object(
MONITOR,
"send_message",
@ -98,6 +193,11 @@ who_need_help_http_requests_total 999
"check_metrics",
return_value=("up", "parsed", "node-a", {"failure": 5.0}),
),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("disabled", "not configured"),
),
mock.patch.object(
MONITOR,
"send_message",
@ -133,6 +233,11 @@ who_need_help_http_requests_total 999
"check_metrics",
return_value=("down", "HTTP 401", "unknown", {}),
),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("disabled", "not configured"),
),
mock.patch.object(
MONITOR,
"send_message",
@ -153,6 +258,75 @@ who_need_help_http_requests_total 999
"check_metrics",
return_value=("up", "parsed", "node-a", {}),
),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("disabled", "not configured"),
),
mock.patch.object(
MONITOR,
"send_message",
side_effect=lambda _config, subject, body: messages.append((subject, body)),
),
mock.patch("builtins.print"),
):
self.assertEqual(MONITOR.monitor(config_path, state_path), 0)
self.assertEqual(len(messages), 2)
self.assertIn("recovered", messages[1][0])
def test_monitor_alerts_for_stale_backup_then_recovers(self):
with tempfile.TemporaryDirectory() as directory:
config_path = Path(directory) / "config.json"
state_path = Path(directory) / "state.json"
config_path.write_text(
json.dumps(
{
"health_url": "https://example.test/healthz/ready",
"metrics_url": "https://example.test/metrics",
}
),
encoding="utf-8",
)
messages = []
with (
mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")),
mock.patch.object(
MONITOR,
"check_metrics",
return_value=("up", "parsed", "node-a", {}),
),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("down", "latest backup is stale"),
),
mock.patch.object(
MONITOR,
"send_message",
side_effect=lambda _config, subject, body: messages.append((subject, body)),
),
mock.patch("builtins.print"),
):
self.assertEqual(MONITOR.monitor(config_path, state_path), 1)
self.assertEqual(len(messages), 1)
self.assertIn("DOWN", messages[0][0])
self.assertIn("backup=down", messages[0][1])
with (
mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")),
mock.patch.object(
MONITOR,
"check_metrics",
return_value=("up", "parsed", "node-a", {}),
),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("up", "latest backup is fresh"),
),
mock.patch.object(
MONITOR,
"send_message",