Monitor restore-verified backup freshness
This commit is contained in:
parent
0a575a6b33
commit
346b03695f
|
|
@ -680,6 +680,12 @@ volume on success, failure, or interrupt. Non-secret catalogs, checksums, and
|
||||||
restore observations remain under ignored
|
restore observations remain under ignored
|
||||||
`output/production-operations/<run-id>/`.
|
`output/production-operations/<run-id>/`.
|
||||||
|
|
||||||
|
After all of those checks pass, `run` atomically publishes a mode-`0600`
|
||||||
|
heartbeat on the independent repository host. The heartbeat contains only the
|
||||||
|
Restic snapshot identifier, verification time, and `restore_verified: true`;
|
||||||
|
it contains no database rows, account identifiers, recipient addresses, or
|
||||||
|
credentials. A failed backup or failed restore never advances the heartbeat.
|
||||||
|
|
||||||
Install the daily user-systemd timer only after one manual `run` succeeds:
|
Install the daily user-systemd timer only after one manual `run` succeeds:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
|
@ -712,6 +718,21 @@ configuration. Readiness or metrics-scrape failure changes the monitor state to
|
||||||
down. Email is sent once when that state changes to down and once when it
|
down. Email is sent once when that state changes to down and once when it
|
||||||
recovers; repeated down checks do not send repeated alerts.
|
recovers; repeated down checks do not send repeated alerts.
|
||||||
|
|
||||||
|
The monitor can also treat a missing, invalid, unverified, or stale backup
|
||||||
|
heartbeat as down. This is intentionally not enabled with an invented default:
|
||||||
|
the maximum acceptable age is the operator's RPO/alerting decision. After that
|
||||||
|
decision is recorded, reinstall the monitor with the chosen positive number of
|
||||||
|
seconds:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
MONITOR_BACKUP_MAX_AGE_SECONDS=<operator-selected-seconds> \
|
||||||
|
./scripts/install-production-external-monitor.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
The installed configuration then records the exact threshold and heartbeat
|
||||||
|
path. A stale backup produces one state-transition alert; a later successful,
|
||||||
|
restore-verified backup produces one recovery notification.
|
||||||
|
|
||||||
For each observed application node, the monitor keeps a baseline of aggregate
|
For each observed application node, the monitor keeps a baseline of aggregate
|
||||||
HTTP exception, failed Oban job, and failed/exceptional email-delivery counters.
|
HTTP exception, failed Oban job, and failed/exceptional email-delivery counters.
|
||||||
It sends one aggregate notification when one of those counters increases. A
|
It sends one aggregate notification when one of those counters increases. A
|
||||||
|
|
|
||||||
|
|
@ -56,6 +56,7 @@ OFFSITE_RESTIC_REPOSITORY=sftp:$repository_target:backups/who_need_help-producti
|
||||||
OFFSITE_RESTIC_PASSWORD_FILE=$password_file
|
OFFSITE_RESTIC_PASSWORD_FILE=$password_file
|
||||||
OFFSITE_RESTIC_HOST=who-need-help-production
|
OFFSITE_RESTIC_HOST=who-need-help-production
|
||||||
OFFSITE_RESTIC_TAG=who-need-help-production
|
OFFSITE_RESTIC_TAG=who-need-help-production
|
||||||
|
OFFSITE_BACKUP_HEARTBEAT_PATH=/home/simple/.local/state/who-need-help/production-backup.json
|
||||||
BACKUP_ON_CALENDAR=daily
|
BACKUP_ON_CALENDAR=daily
|
||||||
EOF
|
EOF
|
||||||
chmod 600 "$config"
|
chmod 600 "$config"
|
||||||
|
|
|
||||||
|
|
@ -16,6 +16,8 @@ metrics_url=${MONITOR_METRICS_URL:-https://whoneedhelp.com/metrics}
|
||||||
monitor_calendar=${MONITOR_ON_CALENDAR:-'*:0/1'}
|
monitor_calendar=${MONITOR_ON_CALENDAR:-'*:0/1'}
|
||||||
health_timeout=${MONITOR_HEALTH_TIMEOUT_SECONDS:-3}
|
health_timeout=${MONITOR_HEALTH_TIMEOUT_SECONDS:-3}
|
||||||
metrics_timeout=${MONITOR_METRICS_TIMEOUT_SECONDS:-3}
|
metrics_timeout=${MONITOR_METRICS_TIMEOUT_SECONDS:-3}
|
||||||
|
backup_heartbeat_path=${MONITOR_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json}
|
||||||
|
backup_max_age=${MONITOR_BACKUP_MAX_AGE_SECONDS:-}
|
||||||
|
|
||||||
for command in awk install mktemp python3 realpath scp ssh stat; do
|
for command in awk install mktemp python3 realpath scp ssh stat; do
|
||||||
command -v "$command" >/dev/null 2>&1 || {
|
command -v "$command" >/dev/null 2>&1 || {
|
||||||
|
|
@ -75,6 +77,17 @@ case "$metrics_timeout" in
|
||||||
0) echo "MONITOR_METRICS_TIMEOUT_SECONDS must be a positive integer." >&2; exit 2 ;;
|
0) echo "MONITOR_METRICS_TIMEOUT_SECONDS must be a positive integer." >&2; exit 2 ;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
|
if [ -n "$backup_max_age" ]; then
|
||||||
|
case "$backup_max_age" in
|
||||||
|
*[!0-9]*) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;;
|
||||||
|
0) echo "MONITOR_BACKUP_MAX_AGE_SECONDS must be a positive integer." >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
if [ "$backup_heartbeat_path" != /home/simple/.local/state/who-need-help/production-backup.json ]; then
|
||||||
|
echo "MONITOR_BACKUP_HEARTBEAT_PATH is outside the reviewed monitor state path." >&2
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX")
|
work_dir=$(mktemp -d "$monitor_install_work_root/monitor-install.XXXXXX")
|
||||||
config="$work_dir/monitor.json"
|
config="$work_dir/monitor.json"
|
||||||
service="$work_dir/who-need-help-production-monitor.service"
|
service="$work_dir/who-need-help-production-monitor.service"
|
||||||
|
|
@ -87,12 +100,21 @@ cleanup() {
|
||||||
trap cleanup 0 HUP INT TERM
|
trap cleanup 0 HUP INT TERM
|
||||||
|
|
||||||
ssh -o BatchMode=yes "$source_target" \
|
ssh -o BatchMode=yes "$source_target" \
|
||||||
"python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source'" >"$config" <<'PY'
|
"python3 - '$production_env' '$monitor_url' '$metrics_url' '$health_timeout' '$metrics_timeout' '$smtp_source' '$backup_heartbeat_path' '$backup_max_age'" >"$config" <<'PY'
|
||||||
import json
|
import json
|
||||||
import shlex
|
import shlex
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
path, health_url, metrics_url, health_timeout, metrics_timeout, smtp_source = sys.argv[1:]
|
(
|
||||||
|
path,
|
||||||
|
health_url,
|
||||||
|
metrics_url,
|
||||||
|
health_timeout,
|
||||||
|
metrics_timeout,
|
||||||
|
smtp_source,
|
||||||
|
backup_heartbeat_path,
|
||||||
|
backup_max_age,
|
||||||
|
) = sys.argv[1:]
|
||||||
smtp_keys = {
|
smtp_keys = {
|
||||||
"SMTP_RELAY",
|
"SMTP_RELAY",
|
||||||
"SMTP_PORT",
|
"SMTP_PORT",
|
||||||
|
|
@ -148,6 +170,11 @@ payload = {
|
||||||
"metrics_url": metrics_url,
|
"metrics_url": metrics_url,
|
||||||
"smtp": smtp,
|
"smtp": smtp,
|
||||||
}
|
}
|
||||||
|
if backup_max_age:
|
||||||
|
payload["backup"] = {
|
||||||
|
"heartbeat_path": backup_heartbeat_path,
|
||||||
|
"max_age_seconds": int(backup_max_age),
|
||||||
|
}
|
||||||
json.dump(payload, sys.stdout, ensure_ascii=False, indent=2, sort_keys=True)
|
json.dump(payload, sys.stdout, ensure_ascii=False, indent=2, sort_keys=True)
|
||||||
sys.stdout.write("\n")
|
sys.stdout.write("\n")
|
||||||
PY
|
PY
|
||||||
|
|
@ -198,6 +225,12 @@ printf 'Health URL: %s\n' "$monitor_url"
|
||||||
printf 'Metrics URL: %s\n' "$metrics_url"
|
printf 'Metrics URL: %s\n' "$metrics_url"
|
||||||
printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \
|
printf 'Schedule: %s; health timeout: %ss; metrics timeout: %ss.\n' \
|
||||||
"$monitor_calendar" "$health_timeout" "$metrics_timeout"
|
"$monitor_calendar" "$health_timeout" "$metrics_timeout"
|
||||||
|
if [ -n "$backup_max_age" ]; then
|
||||||
|
printf 'Backup freshness: heartbeat %s; operator-selected maximum age %ss.\n' \
|
||||||
|
"$backup_heartbeat_path" "$backup_max_age"
|
||||||
|
else
|
||||||
|
echo "Backup freshness monitoring is not enabled because no operator-selected maximum age was supplied."
|
||||||
|
fi
|
||||||
echo "The SMTP and metrics credentials are stored only in a mode-0600 configuration on the external monitor host."
|
echo "The SMTP and metrics credentials are stored only in a mode-0600 configuration on the external monitor host."
|
||||||
if [ -n "$monitor_smtp_values_file" ]; then
|
if [ -n "$monitor_smtp_values_file" ]; then
|
||||||
echo "The external monitor uses the independently supplied SMTP credential set."
|
echo "The external monitor uses the independently supplied SMTP credential set."
|
||||||
|
|
|
||||||
|
|
@ -212,6 +212,64 @@ def check_metrics(config: dict[str, Any]) -> tuple[str, str, str, dict[str, floa
|
||||||
return "down", f"{type(error).__name__}: {str(error)[:500]}", "unknown", {}
|
return "down", f"{type(error).__name__}: {str(error)[:500]}", "unknown", {}
|
||||||
|
|
||||||
|
|
||||||
|
def parse_utc_timestamp(value: Any) -> datetime:
|
||||||
|
if not isinstance(value, str) or not value.strip():
|
||||||
|
raise ValueError("Missing verified backup timestamp")
|
||||||
|
|
||||||
|
parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00"))
|
||||||
|
if parsed.tzinfo is None:
|
||||||
|
raise ValueError("Backup timestamp must include a UTC offset")
|
||||||
|
return parsed.astimezone(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
def check_backup_freshness(
|
||||||
|
config: dict[str, Any], *, now: datetime | None = None
|
||||||
|
) -> tuple[str, str]:
|
||||||
|
backup = config.get("backup")
|
||||||
|
if backup is None:
|
||||||
|
return "disabled", "backup freshness monitoring is not configured"
|
||||||
|
if not isinstance(backup, dict):
|
||||||
|
raise ValueError("Backup monitoring configuration must be an object")
|
||||||
|
|
||||||
|
heartbeat_path = Path(required_text(backup, "heartbeat_path"))
|
||||||
|
if not heartbeat_path.is_absolute():
|
||||||
|
raise ValueError("Backup heartbeat path must be absolute")
|
||||||
|
|
||||||
|
raw_max_age = backup.get("max_age_seconds")
|
||||||
|
if isinstance(raw_max_age, bool):
|
||||||
|
raise ValueError("Backup max_age_seconds must be a positive integer")
|
||||||
|
try:
|
||||||
|
max_age_seconds = int(raw_max_age)
|
||||||
|
except (TypeError, ValueError) as error:
|
||||||
|
raise ValueError("Backup max_age_seconds must be a positive integer") from error
|
||||||
|
if max_age_seconds <= 0:
|
||||||
|
raise ValueError("Backup max_age_seconds must be positive")
|
||||||
|
|
||||||
|
try:
|
||||||
|
heartbeat = read_json(heartbeat_path)
|
||||||
|
if heartbeat.get("restore_verified") is not True:
|
||||||
|
return "down", "latest backup heartbeat is not restore-verified"
|
||||||
|
required_text(heartbeat, "snapshot_id")
|
||||||
|
verified_at = parse_utc_timestamp(heartbeat.get("verified_at"))
|
||||||
|
except (OSError, ValueError, json.JSONDecodeError) as error:
|
||||||
|
return "down", f"{type(error).__name__}: {str(error)[:500]}"
|
||||||
|
|
||||||
|
observed_at = (now or datetime.now(timezone.utc)).astimezone(timezone.utc)
|
||||||
|
if verified_at > observed_at:
|
||||||
|
return "down", "latest backup heartbeat timestamp is in the future"
|
||||||
|
|
||||||
|
age_seconds = max(0, int((observed_at - verified_at).total_seconds()))
|
||||||
|
if age_seconds > max_age_seconds:
|
||||||
|
return (
|
||||||
|
"down",
|
||||||
|
f"latest restore-verified backup is stale (age={age_seconds}s; max={max_age_seconds}s)",
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
"up",
|
||||||
|
f"latest restore-verified backup is fresh (age={age_seconds}s; max={max_age_seconds}s)",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def monitor(config_path: Path, state_path: Path) -> int:
|
def monitor(config_path: Path, state_path: Path) -> int:
|
||||||
config = read_json(config_path)
|
config = read_json(config_path)
|
||||||
previous: dict[str, Any] = {}
|
previous: dict[str, Any] = {}
|
||||||
|
|
@ -220,8 +278,21 @@ def monitor(config_path: Path, state_path: Path) -> int:
|
||||||
|
|
||||||
health_status, health_detail = check_health(config)
|
health_status, health_detail = check_health(config)
|
||||||
metrics_status, metrics_detail, metrics_node, counters = check_metrics(config)
|
metrics_status, metrics_detail, metrics_node, counters = check_metrics(config)
|
||||||
status = "up" if health_status == "up" and metrics_status == "up" else "down"
|
backup_status, backup_detail = check_backup_freshness(config)
|
||||||
detail = f"readiness={health_status} ({health_detail}); metrics={metrics_status} ({metrics_detail})"
|
status = (
|
||||||
|
"up"
|
||||||
|
if health_status == "up"
|
||||||
|
and metrics_status == "up"
|
||||||
|
and backup_status in {"up", "disabled"}
|
||||||
|
else "down"
|
||||||
|
)
|
||||||
|
detail = "; ".join(
|
||||||
|
[
|
||||||
|
f"readiness={health_status} ({health_detail})",
|
||||||
|
f"metrics={metrics_status} ({metrics_detail})",
|
||||||
|
f"backup={backup_status} ({backup_detail})",
|
||||||
|
]
|
||||||
|
)
|
||||||
previous_status = previous.get("status")
|
previous_status = previous.get("status")
|
||||||
|
|
||||||
if status != previous_status:
|
if status != previous_status:
|
||||||
|
|
@ -231,7 +302,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
|
||||||
"[Who Need Help] Production monitor is DOWN",
|
"[Who Need Help] Production monitor is DOWN",
|
||||||
"\n".join(
|
"\n".join(
|
||||||
[
|
[
|
||||||
"The independent production readiness or metrics check failed.",
|
"The independent production readiness, metrics, or backup-freshness check failed.",
|
||||||
f"URL: {required_text(config, 'health_url')}",
|
f"URL: {required_text(config, 'health_url')}",
|
||||||
f"Metrics URL: {required_text(config, 'metrics_url')}",
|
f"Metrics URL: {required_text(config, 'metrics_url')}",
|
||||||
f"Observed at: {utc_now()}",
|
f"Observed at: {utc_now()}",
|
||||||
|
|
@ -246,7 +317,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
|
||||||
"[Who Need Help] Production monitor recovered",
|
"[Who Need Help] Production monitor recovered",
|
||||||
"\n".join(
|
"\n".join(
|
||||||
[
|
[
|
||||||
"The independent production readiness and metrics checks recovered.",
|
"The independent production readiness, metrics, and configured backup-freshness checks recovered.",
|
||||||
f"URL: {required_text(config, 'health_url')}",
|
f"URL: {required_text(config, 'health_url')}",
|
||||||
f"Metrics URL: {required_text(config, 'metrics_url')}",
|
f"Metrics URL: {required_text(config, 'metrics_url')}",
|
||||||
f"Observed at: {utc_now()}",
|
f"Observed at: {utc_now()}",
|
||||||
|
|
@ -286,6 +357,7 @@ def monitor(config_path: Path, state_path: Path) -> int:
|
||||||
state_path,
|
state_path,
|
||||||
{
|
{
|
||||||
"checked_at": utc_now(),
|
"checked_at": utc_now(),
|
||||||
|
"backup_status": backup_status,
|
||||||
"detail": detail,
|
"detail": detail,
|
||||||
"health_status": health_status,
|
"health_status": health_status,
|
||||||
"metrics_by_node": metrics_by_node,
|
"metrics_by_node": metrics_by_node,
|
||||||
|
|
|
||||||
|
|
@ -40,6 +40,7 @@ source "$config"
|
||||||
: "${OFFSITE_RESTIC_PASSWORD_FILE:?Set OFFSITE_RESTIC_PASSWORD_FILE}"
|
: "${OFFSITE_RESTIC_PASSWORD_FILE:?Set OFFSITE_RESTIC_PASSWORD_FILE}"
|
||||||
: "${OFFSITE_RESTIC_HOST:?Set OFFSITE_RESTIC_HOST}"
|
: "${OFFSITE_RESTIC_HOST:?Set OFFSITE_RESTIC_HOST}"
|
||||||
: "${OFFSITE_RESTIC_TAG:?Set OFFSITE_RESTIC_TAG}"
|
: "${OFFSITE_RESTIC_TAG:?Set OFFSITE_RESTIC_TAG}"
|
||||||
|
OFFSITE_BACKUP_HEARTBEAT_PATH=${OFFSITE_BACKUP_HEARTBEAT_PATH:-/home/simple/.local/state/who-need-help/production-backup.json}
|
||||||
|
|
||||||
if [[ "$PRODUCTION_EXPECTED_ENVIRONMENT" != production ]]; then
|
if [[ "$PRODUCTION_EXPECTED_ENVIRONMENT" != production ]]; then
|
||||||
echo "The source environment must be exactly production." >&2
|
echo "The source environment must be exactly production." >&2
|
||||||
|
|
@ -62,6 +63,11 @@ case "$OFFSITE_RESTIC_REPOSITORY" in
|
||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
|
if [[ "$OFFSITE_BACKUP_HEARTBEAT_PATH" != /home/simple/.local/state/who-need-help/production-backup.json ]]; then
|
||||||
|
echo "The backup heartbeat path is outside the reviewed independent-monitor state path." >&2
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
|
||||||
if [[ ! -f "$OFFSITE_RESTIC_PASSWORD_FILE" ]] ||
|
if [[ ! -f "$OFFSITE_RESTIC_PASSWORD_FILE" ]] ||
|
||||||
[[ "$(stat -c '%a' "$OFFSITE_RESTIC_PASSWORD_FILE")" != 600 ]]; then
|
[[ "$(stat -c '%a' "$OFFSITE_RESTIC_PASSWORD_FILE")" != 600 ]]; then
|
||||||
echo "The Restic password file must exist with mode 0600." >&2
|
echo "The Restic password file must exist with mode 0600." >&2
|
||||||
|
|
@ -182,6 +188,8 @@ restore_container="wnh-production-restore-$run_id"
|
||||||
restore_volume="wnh_production_restore_${run_id//[^a-zA-Z0-9]/_}"
|
restore_volume="wnh_production_restore_${run_id//[^a-zA-Z0-9]/_}"
|
||||||
restore_started=false
|
restore_started=false
|
||||||
restore_volume_created=false
|
restore_volume_created=false
|
||||||
|
heartbeat_staged=false
|
||||||
|
heartbeat_remote_tmp="$OFFSITE_BACKUP_HEARTBEAT_PATH.tmp-$run_id"
|
||||||
|
|
||||||
mkdir -p "$evidence_dir"
|
mkdir -p "$evidence_dir"
|
||||||
chmod 700 "$ROOT/output" "$ROOT/output/production-operations" "$evidence_dir" "$work_dir"
|
chmod 700 "$ROOT/output" "$ROOT/output/production-operations" "$evidence_dir" "$work_dir"
|
||||||
|
|
@ -199,6 +207,10 @@ cleanup() {
|
||||||
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" \
|
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'" \
|
||||||
>/dev/null 2>&1 || true
|
>/dev/null 2>&1 || true
|
||||||
fi
|
fi
|
||||||
|
if [[ "$heartbeat_staged" == true ]]; then
|
||||||
|
ssh -o BatchMode=yes "$repository_alias" \
|
||||||
|
"rm -f -- '$heartbeat_remote_tmp'" >/dev/null 2>&1 || true
|
||||||
|
fi
|
||||||
rm -rf "$work_dir"
|
rm -rf "$work_dir"
|
||||||
}
|
}
|
||||||
trap cleanup EXIT HUP INT TERM
|
trap cleanup EXIT HUP INT TERM
|
||||||
|
|
@ -379,6 +391,21 @@ ssh -o BatchMode=yes "$PRODUCTION_SSH_TARGET" \
|
||||||
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'"
|
"rm -f -- '$remote_dump' '$remote_dump.sha256' '$remote_dump.metadata'"
|
||||||
remote_created=false
|
remote_created=false
|
||||||
|
|
||||||
|
heartbeat="$work_dir/production-backup-heartbeat.json"
|
||||||
|
jq -n \
|
||||||
|
--arg snapshot_id "$snapshot_id" \
|
||||||
|
--arg verified_at "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \
|
||||||
|
'{snapshot_id: $snapshot_id, verified_at: $verified_at, restore_verified: true}' \
|
||||||
|
>"$heartbeat"
|
||||||
|
chmod 600 "$heartbeat"
|
||||||
|
ssh -o BatchMode=yes "$repository_alias" \
|
||||||
|
"install -d -m 700 '$(dirname -- "$OFFSITE_BACKUP_HEARTBEAT_PATH")'"
|
||||||
|
heartbeat_staged=true
|
||||||
|
scp -q "$heartbeat" "$repository_alias:$heartbeat_remote_tmp"
|
||||||
|
ssh -o BatchMode=yes "$repository_alias" \
|
||||||
|
"chmod 600 '$heartbeat_remote_tmp' && mv -f -- '$heartbeat_remote_tmp' '$OFFSITE_BACKUP_HEARTBEAT_PATH'"
|
||||||
|
heartbeat_staged=false
|
||||||
|
|
||||||
echo "Encrypted off-server backup and isolated restore drill passed."
|
echo "Encrypted off-server backup and isolated restore drill passed."
|
||||||
printf 'Snapshot: %s\n' "$snapshot_id"
|
printf 'Snapshot: %s\n' "$snapshot_id"
|
||||||
printf 'Non-secret evidence: %s\n' "$evidence_dir"
|
printf 'Non-secret evidence: %s\n' "$evidence_dir"
|
||||||
|
|
|
||||||
|
|
@ -116,7 +116,7 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
def run_installer(self, *, override):
|
def run_installer(self, *, override, backup_max_age=None):
|
||||||
self.capture.unlink(missing_ok=True)
|
self.capture.unlink(missing_ok=True)
|
||||||
env = os.environ.copy()
|
env = os.environ.copy()
|
||||||
env.update(
|
env.update(
|
||||||
|
|
@ -133,6 +133,10 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
||||||
env["MONITOR_SMTP_VALUES_FILE"] = str(self.monitor_values)
|
env["MONITOR_SMTP_VALUES_FILE"] = str(self.monitor_values)
|
||||||
else:
|
else:
|
||||||
env.pop("MONITOR_SMTP_VALUES_FILE", None)
|
env.pop("MONITOR_SMTP_VALUES_FILE", None)
|
||||||
|
if backup_max_age is not None:
|
||||||
|
env["MONITOR_BACKUP_MAX_AGE_SECONDS"] = str(backup_max_age)
|
||||||
|
else:
|
||||||
|
env.pop("MONITOR_BACKUP_MAX_AGE_SECONDS", None)
|
||||||
return subprocess.run(
|
return subprocess.run(
|
||||||
[str(INSTALLER)],
|
[str(INSTALLER)],
|
||||||
cwd=ROOT,
|
cwd=ROOT,
|
||||||
|
|
@ -165,6 +169,27 @@ class InstallProductionExternalMonitorTest(unittest.TestCase):
|
||||||
self.assertNotIn("application-secret", result.stdout + result.stderr)
|
self.assertNotIn("application-secret", result.stdout + result.stderr)
|
||||||
self.assertIn("mirrors the production application SMTP", result.stdout)
|
self.assertIn("mirrors the production application SMTP", result.stdout)
|
||||||
|
|
||||||
|
def test_operator_selected_backup_age_reaches_monitor_config(self):
|
||||||
|
result = self.run_installer(override=True, backup_max_age=172800)
|
||||||
|
|
||||||
|
self.assertEqual(result.returncode, 0, result.stderr)
|
||||||
|
captured = json.loads(self.capture.read_text(encoding="utf-8"))
|
||||||
|
self.assertEqual(
|
||||||
|
captured["backup"],
|
||||||
|
{
|
||||||
|
"heartbeat_path": "/home/simple/.local/state/who-need-help/production-backup.json",
|
||||||
|
"max_age_seconds": 172800,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
self.assertIn("operator-selected maximum age 172800s", result.stdout)
|
||||||
|
|
||||||
|
def test_invalid_backup_age_fails_before_external_changes(self):
|
||||||
|
result = self.run_installer(override=True, backup_max_age="tomorrow")
|
||||||
|
|
||||||
|
self.assertEqual(result.returncode, 2)
|
||||||
|
self.assertFalse(self.capture.exists())
|
||||||
|
self.assertIn("must be a positive integer", result.stderr)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|
|
||||||
|
|
@ -4,6 +4,7 @@ import importlib.util
|
||||||
import json
|
import json
|
||||||
import tempfile
|
import tempfile
|
||||||
import unittest
|
import unittest
|
||||||
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from unittest import mock
|
from unittest import mock
|
||||||
|
|
||||||
|
|
@ -16,6 +17,95 @@ SPEC.loader.exec_module(MONITOR)
|
||||||
|
|
||||||
|
|
||||||
class ProductionExternalMonitorTest(unittest.TestCase):
|
class ProductionExternalMonitorTest(unittest.TestCase):
|
||||||
|
def test_backup_freshness_is_disabled_without_an_operator_policy(self):
|
||||||
|
self.assertEqual(
|
||||||
|
MONITOR.check_backup_freshness({}),
|
||||||
|
("disabled", "backup freshness monitoring is not configured"),
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_backup_freshness_rejects_missing_or_boolean_age(self):
|
||||||
|
for value in (None, True):
|
||||||
|
with self.subTest(value=value):
|
||||||
|
with self.assertRaisesRegex(ValueError, "must be a positive integer"):
|
||||||
|
MONITOR.check_backup_freshness(
|
||||||
|
{
|
||||||
|
"backup": {
|
||||||
|
"heartbeat_path": "/tmp/backup.json",
|
||||||
|
"max_age_seconds": value,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_backup_freshness_accepts_recent_restore_verified_heartbeat(self):
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
heartbeat = Path(directory) / "backup.json"
|
||||||
|
heartbeat.write_text(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"restore_verified": True,
|
||||||
|
"snapshot_id": "abc123",
|
||||||
|
"verified_at": "2026-08-12T00:00:00Z",
|
||||||
|
}
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
status, detail = MONITOR.check_backup_freshness(
|
||||||
|
{
|
||||||
|
"backup": {
|
||||||
|
"heartbeat_path": str(heartbeat),
|
||||||
|
"max_age_seconds": 86_400,
|
||||||
|
}
|
||||||
|
},
|
||||||
|
now=datetime(2026, 8, 12, 12, 0, tzinfo=timezone.utc),
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(status, "up")
|
||||||
|
self.assertIn("age=43200s", detail)
|
||||||
|
|
||||||
|
def test_backup_freshness_rejects_stale_or_unverified_heartbeat(self):
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
heartbeat = Path(directory) / "backup.json"
|
||||||
|
config = {
|
||||||
|
"backup": {
|
||||||
|
"heartbeat_path": str(heartbeat),
|
||||||
|
"max_age_seconds": 3_600,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
heartbeat.write_text(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"restore_verified": True,
|
||||||
|
"snapshot_id": "abc123",
|
||||||
|
"verified_at": "2026-08-12T00:00:00Z",
|
||||||
|
}
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
status, detail = MONITOR.check_backup_freshness(
|
||||||
|
config,
|
||||||
|
now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc),
|
||||||
|
)
|
||||||
|
self.assertEqual(status, "down")
|
||||||
|
self.assertIn("stale", detail)
|
||||||
|
|
||||||
|
heartbeat.write_text(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"restore_verified": False,
|
||||||
|
"snapshot_id": "abc123",
|
||||||
|
"verified_at": "2026-08-12T02:00:00Z",
|
||||||
|
}
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
MONITOR.check_backup_freshness(
|
||||||
|
config,
|
||||||
|
now=datetime(2026, 8, 12, 2, 0, tzinfo=timezone.utc),
|
||||||
|
),
|
||||||
|
("down", "latest backup heartbeat is not restore-verified"),
|
||||||
|
)
|
||||||
|
|
||||||
def test_parses_only_aggregate_failure_counters(self):
|
def test_parses_only_aggregate_failure_counters(self):
|
||||||
payload = """
|
payload = """
|
||||||
# TYPE who_need_help_http_exceptions_total counter
|
# TYPE who_need_help_http_exceptions_total counter
|
||||||
|
|
@ -80,6 +170,11 @@ who_need_help_http_requests_total 999
|
||||||
"check_metrics",
|
"check_metrics",
|
||||||
return_value=("up", "parsed", "node-a", {"failure": 2.0}),
|
return_value=("up", "parsed", "node-a", {"failure": 2.0}),
|
||||||
),
|
),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"check_backup_freshness",
|
||||||
|
return_value=("disabled", "not configured"),
|
||||||
|
),
|
||||||
mock.patch.object(
|
mock.patch.object(
|
||||||
MONITOR,
|
MONITOR,
|
||||||
"send_message",
|
"send_message",
|
||||||
|
|
@ -98,6 +193,11 @@ who_need_help_http_requests_total 999
|
||||||
"check_metrics",
|
"check_metrics",
|
||||||
return_value=("up", "parsed", "node-a", {"failure": 5.0}),
|
return_value=("up", "parsed", "node-a", {"failure": 5.0}),
|
||||||
),
|
),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"check_backup_freshness",
|
||||||
|
return_value=("disabled", "not configured"),
|
||||||
|
),
|
||||||
mock.patch.object(
|
mock.patch.object(
|
||||||
MONITOR,
|
MONITOR,
|
||||||
"send_message",
|
"send_message",
|
||||||
|
|
@ -133,6 +233,11 @@ who_need_help_http_requests_total 999
|
||||||
"check_metrics",
|
"check_metrics",
|
||||||
return_value=("down", "HTTP 401", "unknown", {}),
|
return_value=("down", "HTTP 401", "unknown", {}),
|
||||||
),
|
),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"check_backup_freshness",
|
||||||
|
return_value=("disabled", "not configured"),
|
||||||
|
),
|
||||||
mock.patch.object(
|
mock.patch.object(
|
||||||
MONITOR,
|
MONITOR,
|
||||||
"send_message",
|
"send_message",
|
||||||
|
|
@ -153,6 +258,75 @@ who_need_help_http_requests_total 999
|
||||||
"check_metrics",
|
"check_metrics",
|
||||||
return_value=("up", "parsed", "node-a", {}),
|
return_value=("up", "parsed", "node-a", {}),
|
||||||
),
|
),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"check_backup_freshness",
|
||||||
|
return_value=("disabled", "not configured"),
|
||||||
|
),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"send_message",
|
||||||
|
side_effect=lambda _config, subject, body: messages.append((subject, body)),
|
||||||
|
),
|
||||||
|
mock.patch("builtins.print"),
|
||||||
|
):
|
||||||
|
self.assertEqual(MONITOR.monitor(config_path, state_path), 0)
|
||||||
|
|
||||||
|
self.assertEqual(len(messages), 2)
|
||||||
|
self.assertIn("recovered", messages[1][0])
|
||||||
|
|
||||||
|
def test_monitor_alerts_for_stale_backup_then_recovers(self):
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
config_path = Path(directory) / "config.json"
|
||||||
|
state_path = Path(directory) / "state.json"
|
||||||
|
config_path.write_text(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"health_url": "https://example.test/healthz/ready",
|
||||||
|
"metrics_url": "https://example.test/metrics",
|
||||||
|
}
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
messages = []
|
||||||
|
|
||||||
|
with (
|
||||||
|
mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"check_metrics",
|
||||||
|
return_value=("up", "parsed", "node-a", {}),
|
||||||
|
),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"check_backup_freshness",
|
||||||
|
return_value=("down", "latest backup is stale"),
|
||||||
|
),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"send_message",
|
||||||
|
side_effect=lambda _config, subject, body: messages.append((subject, body)),
|
||||||
|
),
|
||||||
|
mock.patch("builtins.print"),
|
||||||
|
):
|
||||||
|
self.assertEqual(MONITOR.monitor(config_path, state_path), 1)
|
||||||
|
|
||||||
|
self.assertEqual(len(messages), 1)
|
||||||
|
self.assertIn("DOWN", messages[0][0])
|
||||||
|
self.assertIn("backup=down", messages[0][1])
|
||||||
|
|
||||||
|
with (
|
||||||
|
mock.patch.object(MONITOR, "check_health", return_value=("up", "ready")),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"check_metrics",
|
||||||
|
return_value=("up", "parsed", "node-a", {}),
|
||||||
|
),
|
||||||
|
mock.patch.object(
|
||||||
|
MONITOR,
|
||||||
|
"check_backup_freshness",
|
||||||
|
return_value=("up", "latest backup is fresh"),
|
||||||
|
),
|
||||||
mock.patch.object(
|
mock.patch.object(
|
||||||
MONITOR,
|
MONITOR,
|
||||||
"send_message",
|
"send_message",
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue
Block a user