Monitor restore-verified backup freshness

This commit is contained in:
SimpleTest 2026-08-12 18:18:29 +03:00
parent 0a575a6b33
commit 346b03695f
7 changed files with 360 additions and 7 deletions

View File

@ -680,6 +680,12 @@ volume on success, failure, or interrupt. Non-secret catalogs, checksums, and
restore observations remain under ignored restore observations remain under ignored
`output/production-operations/<run-id>/`. `output/production-operations/<run-id>/`.
After all of those checks pass, `run` atomically publishes a mode-`0600`
heartbeat on the independent repository host. The heartbeat contains only the
Restic snapshot identifier, verification time, and `restore_verified: true`;
it contains no database rows, account identifiers, recipient addresses, or
credentials. A failed backup or failed restore never advances the heartbeat.
Install the daily user-systemd timer only after one manual `run` succeeds: Install the daily user-systemd timer only after one manual `run` succeeds:
```bash ```bash
@ -712,6 +718,21 @@ configuration. Readiness or metrics-scrape failure changes the monitor state to
down. Email is sent once when that state changes to down and once when it down. Email is sent once when that state changes to down and once when it
recovers; repeated down checks do not send repeated alerts. recovers; repeated down checks do not send repeated alerts.
The monitor can also treat a missing, invalid, unverified, or stale backup
heartbeat as down. This is intentionally not enabled with an invented default:
the maximum acceptable age is the operator's RPO/alerting decision. After that
decision is recorded, reinstall the monitor with the chosen positive number of
seconds:
```bash
MONITOR_BACKUP_MAX_AGE_SECONDS=<operator-selected-seconds> \
./scripts/install-production-external-monitor.sh
```
The installed configuration then records the exact threshold and heartbeat
path. A stale backup produces one state-transition alert; a later successful,
restore-verified backup produces one recovery notification.
For each observed application node, the monitor keeps a baseline of aggregate For each observed application node, the monitor keeps a baseline of aggregate
HTTP exception, failed Oban job, and failed/exceptional email-delivery counters. HTTP exception, failed Oban job, and failed/exceptional email-delivery counters.
It sends one aggregate notification when one of those counters increases. A It sends one aggregate notification when one of those counters increases. A

View File

@ -56,6 +56,7 @@ OFFSITE_RESTIC_REPOSITORY=sftp:$repository_target:backups/who_need_help-producti
OFFSITE_RESTIC_PASSWORD_FILE=$password_file OFFSITE_RESTIC_PASSWORD_FILE=$password_file
OFFSITE_RESTIC_HOST=who-need-help-production OFFSITE_RESTIC_HOST=who-need-help-production
OFFSITE_RESTIC_TAG=who-need-help-production OFFSITE_RESTIC_TAG=who-need-help-production
OFFSITE_BACKUP_HEARTBEAT_PATH=/home/simple/.local/state/who-need-help/production-backup.json
BACKUP_ON_CALENDAR=daily BACKUP_ON_CALENDAR=daily
EOF EOF
chmod 600 "$config" chmod 600 "$config"

View File

@ -16,6 +16,8 @@ metrics_url=${MONITOR_METRICS_URL:-https://whoneedhelp.com/metrics}
monitor_calendar=${MONITOR_ON_CALENDAR:-'*:0/1'} monitor_calendar=${MONITOR_ON_CALENDAR:-'*:0/1'}
health_timeout=${MONITOR_HEALTH_TIMEOUT_SECONDS:-3} health_timeout=${MONITOR_HEALTH_TIMEOUT_SECONDS:-3}
metrics_timeout=${MONITOR_METRICS_TIMEOUT_SECONDS:-3} metrics_timeout=${MONITOR_METRICS_TIMEOUT_SECONDS:-3}
backup_heartbeat_path=${MONITOR_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json}
backup_max_age=${MONITOR_BACKUP_MAX_AGE_SECONDS:-}
for command in awk install mktemp python3 realpath scp ssh stat; do for command in awk install mktemp python3 realpath scp ssh stat; do
command -v "$command" >/dev/null 2>&1 || { command -v "$command" >/dev/null 2>&1 || {
@ -75,6 +77,17 @@ case "$metrics_timeout" in
0) echo "MONITOR_METRICS_TIMEOUT_SECONDS must be a positive integer." >&2; exit 2 ;; 0) echo "MONITOR_METRICS_TIMEOUT_SECONDS must be a positive integer." >&2; exit 2 ;;
esac esac
if [ -n "$backup_max_age" ]; then
case "$backup_max_age" in
*[!0-9]*) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;;
0) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;;
esac
if [ "$backup_heartbeat_path" != /home/simple/.local/state/who-need-help/production-backup.json ]; then
echo "MONITOR_BACKUP_HEARTBEAT_PATH is outside the reviewed monitor state path." >&2
exit 2
fi
fi
work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX") work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX")
config="$work_dir/monitor.json" config="$work_dir/monitor.json"
service="$work_dir/who-need-help-production-monitor.service" service="$work_dir/who-need-help-production-monitor.service"
@ -87,12 +100,21 @@ cleanup() {
trap cleanup 0 HUP INT TERM trap cleanup 0 HUP INT TERM
ssh -o BatchMode=yes "$source_target" \ ssh -o BatchMode=yes "$source_target" \
"python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source'" >"$config" <<'PY' "python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source' '$backup_heartbeat_path' '$backup_max_age'" >"$config" <<'PY'
import json import json
import shlex import shlex
import sys import sys
path, health_url, metrics_url, health_timeout, metrics_timeout, smtp_source = sys.argv[1:] (
path,
health_url,
metrics_url,
health_timeout,
metrics_timeout,
smtp_source,
backup_heartbeat_path,
backup_max_age,
) = sys.argv[1:]
smtp_keys = { smtp_keys = {
"SMTP_RELAY", "SMTP_RELAY",
"SMTP_PORT", "SMTP_PORT",
@ -148,6 +170,11 @@ payload = {
"metrics_url": metrics_url, "metrics_url": metrics_url,
"smtp": smtp, "smtp": smtp,
} }
if backup_max_age:
payload["backup"] = {
"heartbeat_path": backup_heartbeat_path,
"max_age_seconds": int(backup_max_age),
}
json.dump(payload, sys.stdout, ensure_ascii=False, indent=2, sort_keys=True) json.dump(payload, sys.stdout, ensure_ascii=False, indent=2, sort_keys=True)
sys.stdout.write("\n") sys.stdout.write("\n")
PY PY
@ -198,6 +225,12 @@ printf 'Health URL: %s\n' "$monitor_url"
printf 'Metrics URL: %s\n' "$metrics_url" printf 'Metrics URL: %s\n' "$metrics_url"
printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \ printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \
"$monitor_calendar" "$health_timeout" "$metrics_timeout" "$monitor_calendar" "$health_timeout" "$metrics_timeout"
if [ -n "$backup_max_age" ]; then
printf 'Backup freshness: heartbeat %s; operator-selected maximum age %ss.\n' \
"$backup_heartbeat_path" "$backup_max_age"
else
echo "Backup freshness monitoring is not enabled because no operator-selected maximum age was supplied."
fi
echo "The SMTP and metrics credentials are stored only in a mode-0600 configuration on the external monitor host." echo "The SMTP and metrics credentials are stored only in a mode-0600 configuration on the external monitor host."
if [ -n "$monitor_smtp_values_file" ]; then if [ -n "$monitor_smtp_values_file" ]; then
echo "The external monitor uses the independently supplied SMTP credential set." echo "The external monitor uses the independently supplied SMTP credential set."

View File

@ -212,6 +212,64 @@ def check_metrics(config: dict[str, Any]) -> tuple[str, str, str, dict[str, floa
return "down", f"{type(error).__name__}: {str(error)[:500]}", "unknown", {} return "down", f"{type(error).__name__}: {str(error)[:500]}", "unknown", {}
def parse_utc_timestamp(value: Any) -> datetime:
if not isinstance(value, str) or not value.strip():
raise ValueError("Missing verified backup timestamp")
parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00"))
if parsed.tzinfo is None:
raise ValueError("Backup timestamp must include a UTC offset")
return parsed.astimezone(timezone.utc)
def check_backup_freshness(
config: dict[str, Any], *, now: datetime | None = None
) -> tuple[str, str]:
backup = config.get("backup")
if backup is None:
return "disabled", "backup freshness monitoring is not configured"
if not isinstance(backup, dict):
raise ValueError("Backup monitoring configuration must be an object")
heartbeat_path = Path(required_text(backup, "heartbeat_path"))
if not heartbeat_path.is_absolute():
raise ValueError("Backup heartbeat path must be absolute")
raw_max_age = backup.get("max_age_seconds")
if isinstance(raw_max_age, bool):
raise ValueError("Backup max_age_seconds must be a positive integer")
try:
max_age_seconds = int(raw_max_age)
except (TypeError, ValueError) as error:
raise ValueError("Backup max_age_seconds must be a positive integer") from error
if max_age_seconds <= 0:
raise ValueError("Backup max_age_seconds must be positive")
try:
heartbeat = read_json(heartbeat_path)
if heartbeat.get("restore_verified") is not True:
return "down", "latest backup heartbeat is not restore-verified"
required_text(heartbeat, "snapshot_id")
verified_at = parse_utc_timestamp(heartbeat.get("verified_at"))
except (OSError, ValueError, json.JSONDecodeError) as error:
return "down", f"{type(error).__name__}: {str(error)[:500]}"
observed_at = (now or datetime.now(timezone.utc)).astimezone(timezone.utc)
if verified_at > observed_at:
return "down", "latest backup heartbeat timestamp is in the future"
age_seconds = max(0, int((observed_at - verified_at).total_seconds()))
if age_seconds > max_age_seconds:
return (
"down",
f"latest restore-verified backup is stale (age={age_seconds}s; max={max_age_seconds}s)",
)
return (
"up",
f"latest restore-verified backup is fresh (age={age_seconds}s; max={max_age_seconds}s)",
)
def monitor(config_path: Path, state_path: Path) -> int: def monitor(config_path: Path, state_path: Path) -> int:
config = read_json(config_path) config = read_json(config_path)
previous: dict[str, Any] = {} previous: dict[str, Any] = {}
@ -220,8 +278,21 @@ def monitor(config_path: Path, state_path: Path) -> int:
health_status, health_detail = check_health(config) health_status, health_detail = check_health(config)
metrics_status, metrics_detail, metrics_node, counters = check_metrics(config) metrics_status, metrics_detail, metrics_node, counters = check_metrics(config)
status = "up" if health_status == "up" and metrics_status == "up" else "down" backup_status, backup_detail = check_backup_freshness(config)
detail = f"readiness={health_status} ({health_detail}); metrics={metrics_status} ({metrics_detail})" status = (
"up"
if health_status == "up"
and metrics_status == "up"
and backup_status in {"up", "disabled"}
else "down"
)
detail = "; ".join(
[
f"readiness={health_status} ({health_detail})",
f"metrics={metrics_status} ({metrics_detail})",
f"backup={backup_status} ({backup_detail})",
]
)
previous_status = previous.get("status") previous_status = previous.get("status")
if status != previous_status: if status != previous_status:
@ -231,7 +302,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
"[Who Need Help] Production monitor is DOWN", "[Who Need Help] Production monitor is DOWN",
"\n".join( "\n".join(
[ [
"The independent production readiness or metrics check failed.", "The independent production readiness, metrics, or backup-freshness check failed.",
f"URL: {required_text(config, 'health_url')}", f"URL: {required_text(config, 'health_url')}",
f"Metrics URL: {required_text(config, 'metrics_url')}", f"Metrics URL: {required_text(config, 'metrics_url')}",
f"Observed at: {utc_now()}", f"Observed at: {utc_now()}",
@ -246,7 +317,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
"[Who Need Help] Production monitor recovered", "[Who Need Help] Production monitor recovered",
"\n".join( "\n".join(
[ [
"The independent production readiness and metrics checks recovered.", "The independent production readiness, metrics, and configured backup-freshness checks recovered.",
f"URL: {required_text(config, 'health_url')}", f"URL: {required_text(config, 'health_url')}",
f"Metrics URL: {required_text(config, 'metrics_url')}", f"Metrics URL: {required_text(config, 'metrics_url')}",
f"Observed at: {utc_now()}", f"Observed at: {utc_now()}",
@ -286,6 +357,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
state_path, state_path,
{ {
"checked_at": utc_now(), "checked_at": utc_now(),
"backup_status": backup_status,
"detail": detail, "detail": detail,
"health_status": health_status, "health_status": health_status,
"metrics_by_node": metrics_by_node, "metrics_by_node": metrics_by_node,

View File

@ -40,6 +40,7 @@ source "$config"
: "${OFFSITE_RESTIC_PASSWORD_FILE:?Set OFFSITE_RESTIC_PASSWORD_FILE}" : "${OFFSITE_RESTIC_PASSWORD_FILE:?Set OFFSITE_RESTIC_PASSWORD_FILE}"
: "${OFFSITE_RESTIC_HOST:?Set OFFSITE_RESTIC_HOST}" : "${OFFSITE_RESTIC_HOST:?Set OFFSITE_RESTIC_HOST}"
: "${OFFSITE_RESTIC_TAG:?Set OFFSITE_RESTIC_TAG}" : "${OFFSITE_RESTIC_TAG:?Set OFFSITE_RESTIC_TAG}"
OFFSITE_BACKUP_HEARTBEAT_PATH=${OFFSITE_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json}
if [[ "$PRODUCTION_EXPECTED_ENVIRONMENT" != production ]]; then if [[ "$PRODUCTION_EXPECTED_ENVIRONMENT" != production ]]; then
echo "The source environment must be exactly production." >&2 echo "The source environment must be exactly production." >&2
@ -62,6 +63,11 @@ case "$OFFSITE_RESTIC_REPOSITORY" in
;; ;;
esac esac
if [[ "$OFFSITE_BACKUP_HEARTBEAT_PATH" != /home/simple/.local/state/who-need-help/production-backup.json ]]; then
echo "The backup heartbeat path is outside the reviewed independent-monitor state path." >&2
exit 2
fi
if [[ ! -f "$OFFSITE_RESTIC_PASSWORD_FILE" ]] || if [[ ! -f "$OFFSITE_RESTIC_PASSWORD_FILE" ]] ||
[[ "$(stat -c '%a' "$OFFSITE_RESTIC_PASSWORD_FILE")" != 600 ]]; then [[ "$(stat -c '%a' "$OFFSITE_RESTIC_PASSWORD_FILE")" != 600 ]]; then
echo "The Restic password file must exist with mode 0600." >&2 echo "The Restic password file must exist with mode 0600." >&2
@ -182,6 +188,8 @@ restore_container="wnh-production-restore-$run_id"
restore_volume="wnh_production_restore_${run_id//[^a-zA-Z0-9]/_}" restore_volume="wnh_production_restore_${run_id//[^a-zA-Z0-9]/_}"
restore_started=false restore_started=false
restore_volume_created=false restore_volume_created=false
heartbeat_staged=false
heartbeat_remote_tmp="$OFFSITE_BACKUP_HEARTBEAT_PATH.tmp-$run_id"
mkdir -p "$evidence_dir" mkdir -p "$evidence_dir"
chmod 700 "$ROOT/output" "$ROOT/output/production-operations" "$evidence_dir" "$work_dir" chmod 700 "$ROOT/output" "$ROOT/output/production-operations" "$evidence_dir" "$work_dir"
@ -199,6 +207,10 @@ cleanup() {
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" \ "rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" \
>/dev/null 2>&1 || true >/dev/null 2>&1 || true
fi fi
if [[ "$heartbeat_staged" == true ]]; then
ssh -o BatchMode=yes "$repository_alias" \
"rm -f -- '$heartbeat_remote_tmp'" >/dev/null 2>&1 || true
fi
rm -rf "$work_dir" rm -rf "$work_dir"
} }
trap cleanup EXIT HUP INT TERM trap cleanup EXIT HUP INT TERM
@ -379,6 +391,21 @@ ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" "rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'"
remote_created=false remote_created=false
heartbeat="$work_dir/production-backup-heartbeat.json"
jq -n \
--arg snapshot_id "$snapshot_id" \
--arg verified_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \
'{snapshot_id: $snapshot_id, verified_at: $verified_at, restore_verified: true}' \
>"$heartbeat"
chmod 600 "$heartbeat"
ssh -o BatchMode=yes "$repository_alias" \
"install -d -m 700 '$(dirname -- "$OFFSITE_BACKUP_HEARTBEAT_PATH")'"
heartbeat_staged=true
scp -q "$heartbeat" "$repository_alias:$heartbeat_remote_tmp"
ssh -o BatchMode=yes "$repository_alias" \
"chmod 600 '$heartbeat_remote_tmp' && mv -f -- '$heartbeat_remote_tmp' '$OFFSITE_BACKUP_HEARTBEAT_PATH'"
heartbeat_staged=false
echo "Encrypted off-server backup and isolated restore drill passed." echo "Encrypted off-server backup and isolated restore drill passed."
printf 'Snapshot: %s\n' "$snapshot_id" printf 'Snapshot: %s\n' "$snapshot_id"
printf 'Non-secret evidence: %s\n' "$evidence_dir" printf 'Non-secret evidence: %s\n' "$evidence_dir"

View File

@ -116,7 +116,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
), ),
) )
def run_installer(self, *, override): def run_installer(self, *, override, backup_max_age=None):
self.capture.unlink(missing_ok=True) self.capture.unlink(missing_ok=True)
env = os.environ.copy() env = os.environ.copy()
env.update( env.update(
@ -133,6 +133,10 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
env["MONITOR_SMTP_VALUES_FILE"] = str(self.monitor_values) env["MONITOR_SMTP_VALUES_FILE"] = str(self.monitor_values)
else: else:
env.pop("MONITOR_SMTP_VALUES_FILE", None) env.pop("MONITOR_SMTP_VALUES_FILE", None)
if backup_max_age is not None:
env["MONITOR_BACKUP_MAX_AGE_SECONDS"] = str(backup_max_age)
else:
env.pop("MONITOR_BACKUP_MAX_AGE_SECONDS", None)
return subprocess.run( return subprocess.run(
[str(INSTALLER)], [str(INSTALLER)],
cwd=ROOT, cwd=ROOT,
@ -165,6 +169,27 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
self.assertNotIn("application-secret", result.stdout + result.stderr) self.assertNotIn("application-secret", result.stdout + result.stderr)
self.assertIn("mirrors the production application SMTP", result.stdout) self.assertIn("mirrors the production application SMTP", result.stdout)
def test_operator_selected_backup_age_reaches_monitor_config(self):
result = self.run_installer(override=True, backup_max_age=172800)
self.assertEqual(result.returncode, 0, result.stderr)
captured = json.loads(self.capture.read_text(encoding="utf-8"))
self.assertEqual(
captured["backup"],
{
"heartbeat_path": "/home/simple/.local/state/who-need-help/production-backup.json",
"max_age_seconds": 172800,
},
)
self.assertIn("operator-selected maximum age 172800s", result.stdout)
def test_invalid_backup_age_fails_before_external_changes(self):
result = self.run_installer(override=True, backup_max_age="tomorrow")
self.assertEqual(result.returncode, 2)
self.assertFalse(self.capture.exists())
self.assertIn("must be a positive integer", result.stderr)
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()

View File

@ -4,6 +4,7 @@ import importlib.util
import json import json
import tempfile import tempfile
import unittest import unittest
from datetime import datetime, timezone
from pathlib import Path from pathlib import Path
from unittest import mock from unittest import mock
@ -16,6 +17,95 @@ SPEC.loader.exec_module(MONITOR)
class ProductionExternalMonitorTest(unittest.TestCase): class ProductionExternalMonitorTest(unittest.TestCase):
def test_backup_freshness_is_disabled_without_an_operator_policy(self):
self.assertEqual(
MONITOR.check_backup_freshness({}),
("disabled", "backup freshness monitoring is not configured"),
)
def test_backup_freshness_rejects_missing_or_boolean_age(self):
for value in (None, True):
with self.subTest(value=value):
with self.assertRaisesRegex(ValueError, "must be a positive integer"):
MONITOR.check_backup_freshness(
{
"backup": {
"heartbeat_path": "/tmp/backup.json",
"max_age_seconds": value,
}
}
)
def test_backup_freshness_accepts_recent_restore_verified_heartbeat(self):
with tempfile.TemporaryDirectory() as directory:
heartbeat = Path(directory) / "backup.json"
heartbeat.write_text(
json.dumps(
{
"restore_verified": True,
"snapshot_id": "abc123",
"verified_at": "2026-08-12T00:00:00Z",
}
),
encoding="utf-8",
)
status, detail = MONITOR.check_backup_freshness(
{
"backup": {
"heartbeat_path": str(heartbeat),
"max_age_seconds": 86_400,
}
},
now=datetime(2026, 8, 12, 12, 0, tzinfo=timezone.utc),
)
self.assertEqual(status, "up")
self.assertIn("age=43200s", detail)
def test_backup_freshness_rejects_stale_or_unverified_heartbeat(self):
with tempfile.TemporaryDirectory() as directory:
heartbeat = Path(directory) / "backup.json"
config = {
"backup": {
"heartbeat_path": str(heartbeat),
"max_age_seconds": 3_600,
}
}
heartbeat.write_text(
json.dumps(
{
"restore_verified": True,
"snapshot_id": "abc123",
"verified_at": "2026-08-12T00:00:00Z",
}
),
encoding="utf-8",
)
status, detail = MONITOR.check_backup_freshness(
config,
now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc),
)
self.assertEqual(status, "down")
self.assertIn("stale", detail)
heartbeat.write_text(
json.dumps(
{
"restore_verified": False,
"snapshot_id": "abc123",
"verified_at": "2026-08-12T02:00:00Z",
}
),
encoding="utf-8",
)
self.assertEqual(
MONITOR.check_backup_freshness(
config,
now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc),
),
("down", "latest backup heartbeat is not restore-verified"),
)
def test_parses_only_aggregate_failure_counters(self): def test_parses_only_aggregate_failure_counters(self):
payload = """ payload = """
# TYPE who_need_help_http_exceptions_total counter # TYPE who_need_help_http_exceptions_total counter
@ -80,6 +170,11 @@ who_need_help_http_requests_total 999
"check_metrics", "check_metrics",
return_value=("up", "parsed", "node-a", {"failure": 2.0}), return_value=("up", "parsed", "node-a", {"failure": 2.0}),
), ),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("disabled", "not configured"),
),
mock.patch.object( mock.patch.object(
MONITOR, MONITOR,
"send_message", "send_message",
@ -98,6 +193,11 @@ who_need_help_http_requests_total 999
"check_metrics", "check_metrics",
return_value=("up", "parsed", "node-a", {"failure": 5.0}), return_value=("up", "parsed", "node-a", {"failure": 5.0}),
), ),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("disabled", "not configured"),
),
mock.patch.object( mock.patch.object(
MONITOR, MONITOR,
"send_message", "send_message",
@ -133,6 +233,11 @@ who_need_help_http_requests_total 999
"check_metrics", "check_metrics",
return_value=("down", "HTTP 401", "unknown", {}), return_value=("down", "HTTP 401", "unknown", {}),
), ),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("disabled", "not configured"),
),
mock.patch.object( mock.patch.object(
MONITOR, MONITOR,
"send_message", "send_message",
@ -153,6 +258,75 @@ who_need_help_http_requests_total 999
"check_metrics", "check_metrics",
return_value=("up", "parsed", "node-a", {}), return_value=("up", "parsed", "node-a", {}),
), ),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("disabled", "not configured"),
),
mock.patch.object(
MONITOR,
"send_message",
side_effect=lambda _config, subject, body: messages.append((subject, body)),
),
mock.patch("builtins.print"),
):
self.assertEqual(MONITOR.monitor(config_path, state_path), 0)
self.assertEqual(len(messages), 2)
self.assertIn("recovered", messages[1][0])
def test_monitor_alerts_for_stale_backup_then_recovers(self):
with tempfile.TemporaryDirectory() as directory:
config_path = Path(directory) / "config.json"
state_path = Path(directory) / "state.json"
config_path.write_text(
json.dumps(
{
"health_url": "https://example.test/healthz/ready",
"metrics_url": "https://example.test/metrics",
}
),
encoding="utf-8",
)
messages = []
with (
mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")),
mock.patch.object(
MONITOR,
"check_metrics",
return_value=("up", "parsed", "node-a", {}),
),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("down", "latest backup is stale"),
),
mock.patch.object(
MONITOR,
"send_message",
side_effect=lambda _config, subject, body: messages.append((subject, body)),
),
mock.patch("builtins.print"),
):
self.assertEqual(MONITOR.monitor(config_path, state_path), 1)
self.assertEqual(len(messages), 1)
self.assertIn("DOWN", messages[0][0])
self.assertIn("backup=down", messages[0][1])
with (
mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")),
mock.patch.object(
MONITOR,
"check_metrics",
return_value=("up", "parsed", "node-a", {}),
),
mock.patch.object(
MONITOR,
"check_backup_freshness",
return_value=("up", "latest backup is fresh"),
),
mock.patch.object( mock.patch.object(
MONITOR, MONITOR,
"send_message", "send_message",