184 lines
6.1 KiB
Python
Executable File
184 lines
6.1 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Summarize production read-only probe samples without inventing limits."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
from collections import defaultdict
|
|
from decimal import Decimal
|
|
from pathlib import Path
|
|
from statistics import fmean
|
|
|
|
|
|
UNIT_BYTES = {
|
|
"B": Decimal(1),
|
|
"kB": Decimal(1_000),
|
|
"MB": Decimal(1_000_000),
|
|
"GB": Decimal(1_000_000_000),
|
|
"KiB": Decimal(1_024),
|
|
"MiB": Decimal(1_048_576),
|
|
"GiB": Decimal(1_073_741_824),
|
|
}
|
|
|
|
|
|
def parse_size(value: str) -> int:
|
|
match = re.fullmatch(r"([0-9]+(?:\.[0-9]+)?)\s*([A-Za-z]+)", value.strip())
|
|
if match is None:
|
|
raise ValueError(f"invalid Docker size value: {value}")
|
|
number, unit = match.groups()
|
|
try:
|
|
multiplier = UNIT_BYTES[unit]
|
|
except KeyError as error:
|
|
raise ValueError(f"unsupported Docker size unit: {unit}") from error
|
|
return int(Decimal(number) * multiplier)
|
|
|
|
|
|
def parse_memory_usage(value: str) -> int:
|
|
used, separator, _limit = value.partition("/")
|
|
if not separator:
|
|
raise ValueError(f"invalid Docker MemUsage value: {value}")
|
|
return parse_size(used)
|
|
|
|
|
|
def parse_cpu(value: str) -> float:
|
|
if not value.endswith("%"):
|
|
raise ValueError(f"invalid Docker CPUPerc value: {value}")
|
|
return float(value.removesuffix("%"))
|
|
|
|
|
|
def summarize(source: Path) -> dict[str, object]:
|
|
hosts: list[dict[str, object]] = []
|
|
containers: dict[str, list[dict[str, object]]] = defaultdict(list)
|
|
|
|
with source.open(encoding="utf-8") as stream:
|
|
for line_number, line in enumerate(stream, start=1):
|
|
if not line.strip():
|
|
continue
|
|
try:
|
|
row = json.loads(line)
|
|
observed_at = str(row["observed_at"])
|
|
host = row["host"]
|
|
hosts.append(
|
|
{
|
|
"observed_at": observed_at,
|
|
"load_1": float(host["load_1"]),
|
|
"mem_total_kib": int(host["mem_total_kib"]),
|
|
"mem_available_kib": int(host["mem_available_kib"]),
|
|
"host_postgres_rss_kib": int(
|
|
host["host_postgres_rss_kib"]
|
|
),
|
|
}
|
|
)
|
|
|
|
for container in row["containers"]:
|
|
name = str(container["Name"])
|
|
containers[name].append(
|
|
{
|
|
"observed_at": observed_at,
|
|
"cpu_percent": parse_cpu(str(container["CPUPerc"])),
|
|
"memory_bytes": parse_memory_usage(
|
|
str(container["MemUsage"])
|
|
),
|
|
"pids": int(container["PIDs"]),
|
|
}
|
|
)
|
|
except (
|
|
KeyError,
|
|
TypeError,
|
|
ValueError,
|
|
json.JSONDecodeError,
|
|
) as error:
|
|
raise ValueError(f"{source}:{line_number}: {error}") from error
|
|
|
|
if not hosts:
|
|
raise ValueError(f"{source}: no host samples")
|
|
if not containers:
|
|
raise ValueError(f"{source}: no container samples")
|
|
|
|
host_load = [float(row["load_1"]) for row in hosts]
|
|
available_memory = [int(row["mem_available_kib"]) for row in hosts]
|
|
postgres_memory = [int(row["host_postgres_rss_kib"]) for row in hosts]
|
|
container_summaries: list[dict[str, object]] = []
|
|
|
|
for name in sorted(containers):
|
|
rows = containers[name]
|
|
cpu = [float(row["cpu_percent"]) for row in rows]
|
|
memory = [int(row["memory_bytes"]) for row in rows]
|
|
pids = [int(row["pids"]) for row in rows]
|
|
container_summaries.append(
|
|
{
|
|
"name": name,
|
|
"sample_count": len(rows),
|
|
"cpu_percent": {
|
|
"average": fmean(cpu),
|
|
"minimum": min(cpu),
|
|
"maximum": max(cpu),
|
|
},
|
|
"memory_bytes": {
|
|
"average": fmean(memory),
|
|
"minimum": min(memory),
|
|
"maximum": max(memory),
|
|
"first": memory[0],
|
|
"last": memory[-1],
|
|
"first_to_last_delta": memory[-1] - memory[0],
|
|
},
|
|
"pids": {
|
|
"minimum": min(pids),
|
|
"maximum": max(pids),
|
|
"first": pids[0],
|
|
"last": pids[-1],
|
|
},
|
|
}
|
|
)
|
|
|
|
return {
|
|
"schema_version": 1,
|
|
"measurement": (
|
|
"one-second read-only host and docker stats samples collected over SSH"
|
|
),
|
|
"thresholds_applied": False,
|
|
"first_observed_at": str(hosts[0]["observed_at"]),
|
|
"last_observed_at": str(hosts[-1]["observed_at"]),
|
|
"sample_count": len(hosts),
|
|
"host": {
|
|
"load_1": {
|
|
"average": fmean(host_load),
|
|
"minimum": min(host_load),
|
|
"maximum": max(host_load),
|
|
},
|
|
"mem_total_kib": {
|
|
"minimum": min(int(row["mem_total_kib"]) for row in hosts),
|
|
"maximum": max(int(row["mem_total_kib"]) for row in hosts),
|
|
},
|
|
"mem_available_kib": {
|
|
"average": fmean(available_memory),
|
|
"minimum": min(available_memory),
|
|
"maximum": max(available_memory),
|
|
},
|
|
"host_postgres_rss_kib": {
|
|
"average": fmean(postgres_memory),
|
|
"minimum": min(postgres_memory),
|
|
"maximum": max(postgres_memory),
|
|
},
|
|
},
|
|
"containers": container_summaries,
|
|
}
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser()
|
|
parser.add_argument("source", type=Path)
|
|
parser.add_argument("destination", type=Path)
|
|
args = parser.parse_args()
|
|
result = summarize(args.source)
|
|
args.destination.write_text(
|
|
json.dumps(result, indent=2, sort_keys=True) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|