Monitor restore-verified backup freshness
This commit is contained in:
parent
0a575a6b33
commit
346b03695f
|
|
@ -680,6 +680,12 @@ volume on success, failure, or interrupt. Non-secret catalogs, checksums, and
|
|||
restore observations remain under ignored
|
||||
`output/production-operations/<run-id>/`.
|
||||
|
||||
After all of those checks pass, `run` atomically publishes a mode-`0600`
|
||||
heartbeat on the independent repository host. The heartbeat contains only the
|
||||
Restic snapshot identifier, verification time, and `restore_verified: true`;
|
||||
it contains no database rows, account identifiers, recipient addresses, or
|
||||
credentials. A failed backup or failed restore never advances the heartbeat.
|
||||
|
||||
Install the daily user-systemd timer only after one manual `run` succeeds:
|
||||
|
||||
```bash
|
||||
|
|
@ -712,6 +718,21 @@ configuration. Readiness or metrics-scrape failure changes the monitor state to
|
|||
down. Email is sent once when that state changes to down and once when it
|
||||
recovers; repeated down checks do not send repeated alerts.
|
||||
|
||||
The monitor can also treat a missing, invalid, unverified, or stale backup
|
||||
heartbeat as down. This is intentionally not enabled with an invented default:
|
||||
the maximum acceptable age is the operator's RPO/alerting decision. After that
|
||||
decision is recorded, reinstall the monitor with the chosen positive number of
|
||||
seconds:
|
||||
|
||||
```bash
|
||||
MONITOR_BACKUP_MAX_AGE_SECONDS=<operator-selected-seconds> \
|
||||
./scripts/install-production-external-monitor.sh
|
||||
```
|
||||
|
||||
The installed configuration then records the exact threshold and heartbeat
|
||||
path. A stale backup produces one state-transition alert; a later successful,
|
||||
restore-verified backup produces one recovery notification.
|
||||
|
||||
For each observed application node, the monitor keeps a baseline of aggregate
|
||||
HTTP exception, failed Oban job, and failed/exceptional email-delivery counters.
|
||||
It sends one aggregate notification when one of those counters increases. A
|
||||
|
|
|
|||
|
|
@ -56,6 +56,7 @@ OFFSITE_RESTIC_REPOSITORY=sftp:$repository_target:backups/who_need_help-producti
|
|||
OFFSITE_RESTIC_PASSWORD_FILE=$password_file
|
||||
OFFSITE_RESTIC_HOST=who-need-help-production
|
||||
OFFSITE_RESTIC_TAG=who-need-help-production
|
||||
OFFSITE_BACKUP_HEARTBEAT_PATH=/home/simple/.local/state/who-need-help/production-backup.json
|
||||
BACKUP_ON_CALENDAR=daily
|
||||
EOF
|
||||
chmod 600 "$config"
|
||||
|
|
|
|||
|
|
@ -16,6 +16,8 @@ metrics_url=${MONITOR_METRICS_URL:-https://whoneedhelp.com/metrics}
|
|||
monitor_calendar=${MONITOR_ON_CALENDAR:-'*:0/1'}
|
||||
health_timeout=${MONITOR_HEALTH_TIMEOUT_SECONDS:-3}
|
||||
metrics_timeout=${MONITOR_METRICS_TIMEOUT_SECONDS:-3}
|
||||
backup_heartbeat_path=${MONITOR_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json}
|
||||
backup_max_age=${MONITOR_BACKUP_MAX_AGE_SECONDS:-}
|
||||
|
||||
for command in awk install mktemp python3 realpath scp ssh stat; do
|
||||
command -v "$command" >/dev/null 2>&1 || {
|
||||
|
|
@ -75,6 +77,17 @@ case "$metrics_timeout" in
|
|||
0) echo "MONITOR_METRICS_TIMEOUT_SECONDS must be a positive integer." >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
if [ -n "$backup_max_age" ]; then
|
||||
case "$backup_max_age" in
|
||||
*[!0-9]*) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;;
|
||||
0) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;;
|
||||
esac
|
||||
if [ "$backup_heartbeat_path" != /home/simple/.local/state/who-need-help/production-backup.json ]; then
|
||||
echo "MONITOR_BACKUP_HEARTBEAT_PATH is outside the reviewed monitor state path." >&2
|
||||
exit 2
|
||||
fi
|
||||
fi
|
||||
|
||||
work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX")
|
||||
config="$work_dir/monitor.json"
|
||||
service="$work_dir/who-need-help-production-monitor.service"
|
||||
|
|
@ -87,12 +100,21 @@ cleanup() {
|
|||
trap cleanup 0 HUP INT TERM
|
||||
|
||||
ssh -o BatchMode=yes "$source_target" \
|
||||
"python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source'" >"$config" <<'PY'
|
||||
"python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source' '$backup_heartbeat_path' '$backup_max_age'" >"$config" <<'PY'
|
||||
import json
|
||||
import shlex
|
||||
import sys
|
||||
|
||||
path, health_url, metrics_url, health_timeout, metrics_timeout, smtp_source = sys.argv[1:]
|
||||
(
|
||||
path,
|
||||
health_url,
|
||||
metrics_url,
|
||||
health_timeout,
|
||||
metrics_timeout,
|
||||
smtp_source,
|
||||
backup_heartbeat_path,
|
||||
backup_max_age,
|
||||
) = sys.argv[1:]
|
||||
smtp_keys = {
|
||||
"SMTP_RELAY",
|
||||
"SMTP_PORT",
|
||||
|
|
@ -148,6 +170,11 @@ payload = {
|
|||
"metrics_url": metrics_url,
|
||||
"smtp": smtp,
|
||||
}
|
||||
if backup_max_age:
|
||||
payload["backup"] = {
|
||||
"heartbeat_path": backup_heartbeat_path,
|
||||
"max_age_seconds": int(backup_max_age),
|
||||
}
|
||||
json.dump(payload, sys.stdout, ensure_ascii=False, indent=2, sort_keys=True)
|
||||
sys.stdout.write("\n")
|
||||
PY
|
||||
|
|
@ -198,6 +225,12 @@ printf 'Health URL: %s\n' "$monitor_url"
|
|||
printf 'Metrics URL: %s\n' "$metrics_url"
|
||||
printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \
|
||||
"$monitor_calendar" "$health_timeout" "$metrics_timeout"
|
||||
if [ -n "$backup_max_age" ]; then
|
||||
printf 'Backup freshness: heartbeat %s; operator-selected maximum age %ss.\n' \
|
||||
"$backup_heartbeat_path" "$backup_max_age"
|
||||
else
|
||||
echo "Backup freshness monitoring is not enabled because no operator-selected maximum age was supplied."
|
||||
fi
|
||||
echo "The SMTP and metrics credentials are stored only in a mode-0600 configuration on the external monitor host."
|
||||
if [ -n "$monitor_smtp_values_file" ]; then
|
||||
echo "The external monitor uses the independently supplied SMTP credential set."
|
||||
|
|
|
|||
|
|
@ -212,6 +212,64 @@ def check_metrics(config: dict[str, Any]) -> tuple[str, str, str, dict[str, floa
|
|||
return "down", f"{type(error).__name__}: {str(error)[:500]}", "unknown", {}
|
||||
|
||||
|
||||
def parse_utc_timestamp(value: Any) -> datetime:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ValueError("Missing verified backup timestamp")
|
||||
|
||||
parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00"))
|
||||
if parsed.tzinfo is None:
|
||||
raise ValueError("Backup timestamp must include a UTC offset")
|
||||
return parsed.astimezone(timezone.utc)
|
||||
|
||||
|
||||
def check_backup_freshness(
|
||||
config: dict[str, Any], *, now: datetime | None = None
|
||||
) -> tuple[str, str]:
|
||||
backup = config.get("backup")
|
||||
if backup is None:
|
||||
return "disabled", "backup freshness monitoring is not configured"
|
||||
if not isinstance(backup, dict):
|
||||
raise ValueError("Backup monitoring configuration must be an object")
|
||||
|
||||
heartbeat_path = Path(required_text(backup, "heartbeat_path"))
|
||||
if not heartbeat_path.is_absolute():
|
||||
raise ValueError("Backup heartbeat path must be absolute")
|
||||
|
||||
raw_max_age = backup.get("max_age_seconds")
|
||||
if isinstance(raw_max_age, bool):
|
||||
raise ValueError("Backup max_age_seconds must be a positive integer")
|
||||
try:
|
||||
max_age_seconds = int(raw_max_age)
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError("Backup max_age_seconds must be a positive integer") from error
|
||||
if max_age_seconds <= 0:
|
||||
raise ValueError("Backup max_age_seconds must be positive")
|
||||
|
||||
try:
|
||||
heartbeat = read_json(heartbeat_path)
|
||||
if heartbeat.get("restore_verified") is not True:
|
||||
return "down", "latest backup heartbeat is not restore-verified"
|
||||
required_text(heartbeat, "snapshot_id")
|
||||
verified_at = parse_utc_timestamp(heartbeat.get("verified_at"))
|
||||
except (OSError, ValueError, json.JSONDecodeError) as error:
|
||||
return "down", f"{type(error).__name__}: {str(error)[:500]}"
|
||||
|
||||
observed_at = (now or datetime.now(timezone.utc)).astimezone(timezone.utc)
|
||||
if verified_at > observed_at:
|
||||
return "down", "latest backup heartbeat timestamp is in the future"
|
||||
|
||||
age_seconds = max(0, int((observed_at - verified_at).total_seconds()))
|
||||
if age_seconds > max_age_seconds:
|
||||
return (
|
||||
"down",
|
||||
f"latest restore-verified backup is stale (age={age_seconds}s; max={max_age_seconds}s)",
|
||||
)
|
||||
return (
|
||||
"up",
|
||||
f"latest restore-verified backup is fresh (age={age_seconds}s; max={max_age_seconds}s)",
|
||||
)
|
||||
|
||||
|
||||
def monitor(config_path: Path, state_path: Path) -> int:
|
||||
config = read_json(config_path)
|
||||
previous: dict[str, Any] = {}
|
||||
|
|
@ -220,8 +278,21 @@ def monitor(config_path: Path, state_path: Path) -> int:
|
|||
|
||||
health_status, health_detail = check_health(config)
|
||||
metrics_status, metrics_detail, metrics_node, counters = check_metrics(config)
|
||||
status = "up" if health_status == "up" and metrics_status == "up" else "down"
|
||||
detail = f"readiness={health_status} ({health_detail}); metrics={metrics_status} ({metrics_detail})"
|
||||
backup_status, backup_detail = check_backup_freshness(config)
|
||||
status = (
|
||||
"up"
|
||||
if health_status == "up"
|
||||
and metrics_status == "up"
|
||||
and backup_status in {"up", "disabled"}
|
||||
else "down"
|
||||
)
|
||||
detail = "; ".join(
|
||||
[
|
||||
f"readiness={health_status} ({health_detail})",
|
||||
f"metrics={metrics_status} ({metrics_detail})",
|
||||
f"backup={backup_status} ({backup_detail})",
|
||||
]
|
||||
)
|
||||
previous_status = previous.get("status")
|
||||
|
||||
if status != previous_status:
|
||||
|
|
@ -231,7 +302,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
|
|||
"[Who Need Help] Production monitor is DOWN",
|
||||
"\n".join(
|
||||
[
|
||||
"The independent production readiness or metrics check failed.",
|
||||
"The independent production readiness, metrics, or backup-freshness check failed.",
|
||||
f"URL: {required_text(config, 'health_url')}",
|
||||
f"Metrics URL: {required_text(config, 'metrics_url')}",
|
||||
f"Observed at: {utc_now()}",
|
||||
|
|
@ -246,7 +317,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
|
|||
"[Who Need Help] Production monitor recovered",
|
||||
"\n".join(
|
||||
[
|
||||
"The independent production readiness and metrics checks recovered.",
|
||||
"The independent production readiness, metrics, and configured backup-freshness checks recovered.",
|
||||
f"URL: {required_text(config, 'health_url')}",
|
||||
f"Metrics URL: {required_text(config, 'metrics_url')}",
|
||||
f"Observed at: {utc_now()}",
|
||||
|
|
@ -286,6 +357,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
|
|||
state_path,
|
||||
{
|
||||
"checked_at": utc_now(),
|
||||
"backup_status": backup_status,
|
||||
"detail": detail,
|
||||
"health_status": health_status,
|
||||
"metrics_by_node": metrics_by_node,
|
||||
|
|
|
|||
|
|
@ -40,6 +40,7 @@ source "$config"
|
|||
: "${OFFSITE_RESTIC_PASSWORD_FILE:?Set OFFSITE_RESTIC_PASSWORD_FILE}"
|
||||
: "${OFFSITE_RESTIC_HOST:?Set OFFSITE_RESTIC_HOST}"
|
||||
: "${OFFSITE_RESTIC_TAG:?Set OFFSITE_RESTIC_TAG}"
|
||||
OFFSITE_BACKUP_HEARTBEAT_PATH=${OFFSITE_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json}
|
||||
|
||||
if [[ "$PRODUCTION_EXPECTED_ENVIRONMENT" != production ]]; then
|
||||
echo "The source environment must be exactly production." >&2
|
||||
|
|
@ -62,6 +63,11 @@ case "$OFFSITE_RESTIC_REPOSITORY" in
|
|||
;;
|
||||
esac
|
||||
|
||||
if [[ "$OFFSITE_BACKUP_HEARTBEAT_PATH" != /home/simple/.local/state/who-need-help/production-backup.json ]]; then
|
||||
echo "The backup heartbeat path is outside the reviewed independent-monitor state path." >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
if [[ ! -f "$OFFSITE_RESTIC_PASSWORD_FILE" ]] ||
|
||||
[[ "$(stat -c '%a' "$OFFSITE_RESTIC_PASSWORD_FILE")" != 600 ]]; then
|
||||
echo "The Restic password file must exist with mode 0600." >&2
|
||||
|
|
@ -182,6 +188,8 @@ restore_container="wnh-production-restore-$run_id"
|
|||
restore_volume="wnh_production_restore_${run_id//[^a-zA-Z0-9]/_}"
|
||||
restore_started=false
|
||||
restore_volume_created=false
|
||||
heartbeat_staged=false
|
||||
heartbeat_remote_tmp="$OFFSITE_BACKUP_HEARTBEAT_PATH.tmp-$run_id"
|
||||
|
||||
mkdir -p "$evidence_dir"
|
||||
chmod 700 "$ROOT/output" "$ROOT/output/production-operations" "$evidence_dir" "$work_dir"
|
||||
|
|
@ -199,6 +207,10 @@ cleanup() {
|
|||
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" \
|
||||
>/dev/null 2>&1 || true
|
||||
fi
|
||||
if [[ "$heartbeat_staged" == true ]]; then
|
||||
ssh -o BatchMode=yes "$repository_alias" \
|
||||
"rm -f -- '$heartbeat_remote_tmp'" >/dev/null 2>&1 || true
|
||||
fi
|
||||
rm -rf "$work_dir"
|
||||
}
|
||||
trap cleanup EXIT HUP INT TERM
|
||||
|
|
@ -379,6 +391,21 @@ ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \
|
|||
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'"
|
||||
remote_created=false
|
||||
|
||||
heartbeat="$work_dir/production-backup-heartbeat.json"
|
||||
jq -n \
|
||||
--arg snapshot_id "$snapshot_id" \
|
||||
--arg verified_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \
|
||||
'{snapshot_id: $snapshot_id, verified_at: $verified_at, restore_verified: true}' \
|
||||
>"$heartbeat"
|
||||
chmod 600 "$heartbeat"
|
||||
ssh -o BatchMode=yes "$repository_alias" \
|
||||
"install -d -m 700 '$(dirname -- "$OFFSITE_BACKUP_HEARTBEAT_PATH")'"
|
||||
heartbeat_staged=true
|
||||
scp -q "$heartbeat" "$repository_alias:$heartbeat_remote_tmp"
|
||||
ssh -o BatchMode=yes "$repository_alias" \
|
||||
"chmod 600 '$heartbeat_remote_tmp' && mv -f -- '$heartbeat_remote_tmp' '$OFFSITE_BACKUP_HEARTBEAT_PATH'"
|
||||
heartbeat_staged=false
|
||||
|
||||
echo "Encrypted off-server backup and isolated restore drill passed."
|
||||
printf 'Snapshot: %s\n' "$snapshot_id"
|
||||
printf 'Non-secret evidence: %s\n' "$evidence_dir"
|
||||
|
|
|
|||
|
|
@ -116,7 +116,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
|||
),
|
||||
)
|
||||
|
||||
def run_installer(self, *, override):
|
||||
def run_installer(self, *, override, backup_max_age=None):
|
||||
self.capture.unlink(missing_ok=True)
|
||||
env = os.environ.copy()
|
||||
env.update(
|
||||
|
|
@ -133,6 +133,10 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
|||
env["MONITOR_SMTP_VALUES_FILE"] = str(self.monitor_values)
|
||||
else:
|
||||
env.pop("MONITOR_SMTP_VALUES_FILE", None)
|
||||
if backup_max_age is not None:
|
||||
env["MONITOR_BACKUP_MAX_AGE_SECONDS"] = str(backup_max_age)
|
||||
else:
|
||||
env.pop("MONITOR_BACKUP_MAX_AGE_SECONDS", None)
|
||||
return subprocess.run(
|
||||
[str(INSTALLER)],
|
||||
cwd=ROOT,
|
||||
|
|
@ -165,6 +169,27 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
|||
self.assertNotIn("application-secret", result.stdout + result.stderr)
|
||||
self.assertIn("mirrors the production application SMTP", result.stdout)
|
||||
|
||||
def test_operator_selected_backup_age_reaches_monitor_config(self):
|
||||
result = self.run_installer(override=True, backup_max_age=172800)
|
||||
|
||||
self.assertEqual(result.returncode, 0, result.stderr)
|
||||
captured = json.loads(self.capture.read_text(encoding="utf-8"))
|
||||
self.assertEqual(
|
||||
captured["backup"],
|
||||
{
|
||||
"heartbeat_path": "/home/simple/.local/state/who-need-help/production-backup.json",
|
||||
"max_age_seconds": 172800,
|
||||
},
|
||||
)
|
||||
self.assertIn("operator-selected maximum age 172800s", result.stdout)
|
||||
|
||||
def test_invalid_backup_age_fails_before_external_changes(self):
|
||||
result = self.run_installer(override=True, backup_max_age="tomorrow")
|
||||
|
||||
self.assertEqual(result.returncode, 2)
|
||||
self.assertFalse(self.capture.exists())
|
||||
self.assertIn("must be a positive integer", result.stderr)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ import importlib.util
|
|||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
|
|
@ -16,6 +17,95 @@ SPEC.loader.exec_module(MONITOR)
|
|||
|
||||
|
||||
class ProductionExternalMonitorTest(unittest.TestCase):
|
||||
def test_backup_freshness_is_disabled_without_an_operator_policy(self):
|
||||
self.assertEqual(
|
||||
MONITOR.check_backup_freshness({}),
|
||||
("disabled", "backup freshness monitoring is not configured"),
|
||||
)
|
||||
|
||||
def test_backup_freshness_rejects_missing_or_boolean_age(self):
|
||||
for value in (None, True):
|
||||
with self.subTest(value=value):
|
||||
with self.assertRaisesRegex(ValueError, "must be a positive integer"):
|
||||
MONITOR.check_backup_freshness(
|
||||
{
|
||||
"backup": {
|
||||
"heartbeat_path": "/tmp/backup.json",
|
||||
"max_age_seconds": value,
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
def test_backup_freshness_accepts_recent_restore_verified_heartbeat(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
heartbeat = Path(directory) / "backup.json"
|
||||
heartbeat.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"restore_verified": True,
|
||||
"snapshot_id": "abc123",
|
||||
"verified_at": "2026-08-12T00:00:00Z",
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
status, detail = MONITOR.check_backup_freshness(
|
||||
{
|
||||
"backup": {
|
||||
"heartbeat_path": str(heartbeat),
|
||||
"max_age_seconds": 86_400,
|
||||
}
|
||||
},
|
||||
now=datetime(2026, 8, 12, 12, 0, tzinfo=timezone.utc),
|
||||
)
|
||||
|
||||
self.assertEqual(status, "up")
|
||||
self.assertIn("age=43200s", detail)
|
||||
|
||||
def test_backup_freshness_rejects_stale_or_unverified_heartbeat(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
heartbeat = Path(directory) / "backup.json"
|
||||
config = {
|
||||
"backup": {
|
||||
"heartbeat_path": str(heartbeat),
|
||||
"max_age_seconds": 3_600,
|
||||
}
|
||||
}
|
||||
heartbeat.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"restore_verified": True,
|
||||
"snapshot_id": "abc123",
|
||||
"verified_at": "2026-08-12T00:00:00Z",
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
status, detail = MONITOR.check_backup_freshness(
|
||||
config,
|
||||
now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc),
|
||||
)
|
||||
self.assertEqual(status, "down")
|
||||
self.assertIn("stale", detail)
|
||||
|
||||
heartbeat.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"restore_verified": False,
|
||||
"snapshot_id": "abc123",
|
||||
"verified_at": "2026-08-12T02:00:00Z",
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
self.assertEqual(
|
||||
MONITOR.check_backup_freshness(
|
||||
config,
|
||||
now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc),
|
||||
),
|
||||
("down", "latest backup heartbeat is not restore-verified"),
|
||||
)
|
||||
|
||||
def test_parses_only_aggregate_failure_counters(self):
|
||||
payload = """
|
||||
# TYPE who_need_help_http_exceptions_total counter
|
||||
|
|
@ -80,6 +170,11 @@ who_need_help_http_requests_total 999
|
|||
"check_metrics",
|
||||
return_value=("up", "parsed", "node-a", {"failure": 2.0}),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"check_backup_freshness",
|
||||
return_value=("disabled", "not configured"),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"send_message",
|
||||
|
|
@ -98,6 +193,11 @@ who_need_help_http_requests_total 999
|
|||
"check_metrics",
|
||||
return_value=("up", "parsed", "node-a", {"failure": 5.0}),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"check_backup_freshness",
|
||||
return_value=("disabled", "not configured"),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"send_message",
|
||||
|
|
@ -133,6 +233,11 @@ who_need_help_http_requests_total 999
|
|||
"check_metrics",
|
||||
return_value=("down", "HTTP 401", "unknown", {}),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"check_backup_freshness",
|
||||
return_value=("disabled", "not configured"),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"send_message",
|
||||
|
|
@ -153,6 +258,75 @@ who_need_help_http_requests_total 999
|
|||
"check_metrics",
|
||||
return_value=("up", "parsed", "node-a", {}),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"check_backup_freshness",
|
||||
return_value=("disabled", "not configured"),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"send_message",
|
||||
side_effect=lambda _config, subject, body: messages.append((subject, body)),
|
||||
),
|
||||
mock.patch("builtins.print"),
|
||||
):
|
||||
self.assertEqual(MONITOR.monitor(config_path, state_path), 0)
|
||||
|
||||
self.assertEqual(len(messages), 2)
|
||||
self.assertIn("recovered", messages[1][0])
|
||||
|
||||
def test_monitor_alerts_for_stale_backup_then_recovers(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
config_path = Path(directory) / "config.json"
|
||||
state_path = Path(directory) / "state.json"
|
||||
config_path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"health_url": "https://example.test/healthz/ready",
|
||||
"metrics_url": "https://example.test/metrics",
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
messages = []
|
||||
|
||||
with (
|
||||
mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"check_metrics",
|
||||
return_value=("up", "parsed", "node-a", {}),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"check_backup_freshness",
|
||||
return_value=("down", "latest backup is stale"),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"send_message",
|
||||
side_effect=lambda _config, subject, body: messages.append((subject, body)),
|
||||
),
|
||||
mock.patch("builtins.print"),
|
||||
):
|
||||
self.assertEqual(MONITOR.monitor(config_path, state_path), 1)
|
||||
|
||||
self.assertEqual(len(messages), 1)
|
||||
self.assertIn("DOWN", messages[0][0])
|
||||
self.assertIn("backup=down", messages[0][1])
|
||||
|
||||
with (
|
||||
mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"check_metrics",
|
||||
return_value=("up", "parsed", "node-a", {}),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"check_backup_freshness",
|
||||
return_value=("up", "latest backup is fresh"),
|
||||
),
|
||||
mock.patch.object(
|
||||
MONITOR,
|
||||
"send_message",
|
||||
|
|
|
|||
Loading…
Reference in New Issue
Block a user