From 465b27d673c5c8b1b388abb9c234c8301dd28c43 Mon Sep 17 00:00:00 2001 From: Sanders Date: Wed, 5 Aug 2026 22:28:20 +0300 Subject: [PATCH] ServerMonitor: Dockerized health-check service with Telegram alerts - Pure stdlib Python (3.12), runs on python:3.12-alpine as non-root - Checks: ping, HTTP (expected codes), TCP ports, Netdata alarms + metrics - Netdata v2 compatible: system.cpu (no idle dim), system.ram, disk_space.* discovery - AlertEngine dedup: first alert, reminders, recovery (only after real alert) - Baseline first-run (no alert storm on deploy), atomic state file, --once mode - 20 unit tests passing; verified live against 192.168.0.5 --- .env.example | 6 + .gitignore | 4 + Dockerfile | 18 +++ README.md | 133 ++++++++++++++++ SPEC.md | 127 ++++++++++++++++ checks.py | 330 ++++++++++++++++++++++++++++++++++++++++ config.json | 24 +++ docker-compose.yml | 17 +++ monitor.py | 296 ++++++++++++++++++++++++++++++++++++ notify.py | 173 +++++++++++++++++++++ tests/test_checks.py | 353 +++++++++++++++++++++++++++++++++++++++++++ 11 files changed, 1481 insertions(+) create mode 100644 .env.example create mode 100644 .gitignore create mode 100644 Dockerfile create mode 100644 README.md create mode 100644 SPEC.md create mode 100644 checks.py create mode 100644 config.json create mode 100644 docker-compose.yml create mode 100644 monitor.py create mode 100644 notify.py create mode 100644 tests/test_checks.py diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..c559c92 --- /dev/null +++ b/.env.example @@ -0,0 +1,6 @@ +# Telegram bot for alerts (get from @BotFather). Leave empty for log-only mode. +TELEGRAM_BOT_TOKEN= +# Your chat id (get from @userinfobot) +TELEGRAM_CHAT_ID= +# Check interval in seconds (default 60) +CHECK_INTERVAL_SEC=60 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..e854d58 --- /dev/null +++ b/.gitignore @@ -0,0 +1,4 @@ +__pycache__/ +*.pyc +.env +state.json diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..c0b41c7 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,18 @@ +FROM python:3.12-alpine + +RUN adduser -D -u 10001 app + +WORKDIR /app + +COPY checks.py notify.py monitor.py config.json ./ + +RUN mkdir -p /data && chown -R app:app /data /app + +USER app + +ENV PYTHONUNBUFFERED=1 + +HEALTHCHECK --interval=60s --timeout=10s --retries=3 --start-period=30s \ + CMD python -c "import os,time; p='/data/state.json'; raise SystemExit(0 if os.path.exists(p) and time.time()-os.path.getmtime(p)<300 else 1)" + +CMD ["python", "-u", "/app/monitor.py"] diff --git a/README.md b/README.md new file mode 100644 index 0000000..0e5d963 --- /dev/null +++ b/README.md @@ -0,0 +1,133 @@ +# ServerMonitor + +Docker-контейнер для мониторинга домашнего сервера (Unraid) с удалённого VPS. Периодически проверяет доступность сервисов и шлёт алерты в Telegram при падениях, напоминания, пока сервис не поднимется, и сообщения о восстановлении. + +## Что это + +ServerMonitor — это лёгкий Python-сервис без сторонних зависимостей. Он работает на `python:3.12-alpine`, использует только стандартную библиотеку и предназначен для развёртывания на VPS. Сервис мониторит хост `192.168.0.5` по сети: пингует его, делает HTTP-запросы к веб-приложениям, проверяет TCP-порты и забирает алармы/метрики из Netdata. + +## Как это работает + +``` +┌─────────┐ ping / http / tcp / netdata ┌──────────────┐ +│ VPS │ ───────────────────────────────→│ Unraid 192. │ +│ контейнер│ │ 168.0.5 │ +│ServerMonitor│ │ │ +└────┬────┘ └──────────────┘ + │ + │ Telegram (alert / reminder / recovery) + ▼ +┌─────────────┐ +│ Telegram │ +└─────────────┘ +``` + +1. Контейнер каждые `check_interval_sec` секунд выполняет набор проверок. +2. Если проверка падает — отправляется первый алерт. +3. Пока сервис остаётся недоступным, каждые `reminder_interval_sec` секунд приходит напоминание. +4. Когда сервис восстанавливается — приходит сообщение о recovery. +5. Состояние хранится в `/data/state.json`, поэтому перезапуск контейнера не вызывает повторной волны алертов. + +## Быстрый старт + +```bash +cp .env.example .env +# отредактируй .env: добавь TELEGRAM_BOT_TOKEN и TELEGRAM_CHAT_ID +docker compose up -d --build +``` + +## Как получить токен бота и chat_id + +1. Напишите [@BotFather](https://t.me/BotFather), создайте бота и скопируйте токен в `TELEGRAM_BOT_TOKEN`. +2. Напишите [@userinfobot](https://t.me/userinfobot), получите свой ID и вставьте его в `TELEGRAM_CHAT_ID`. + +## Настройка чеков в config.json + +| Секция | Поле | Описание | +|--------|------|----------| +| `host` | строка | IP или имя мониторимого хоста | +| `check_interval_sec` | число | Интервал между циклами проверок, сек | +| `reminder_interval_sec` | число | Интервал между напоминаниями, сек | +| `timeout_sec` | число | Таймаут сетевых операций, сек | +| `state_file` | строка | Путь к файлу состояния | +| `telegram.bot_token` | строка | Токен Telegram-бота | +| `telegram.chat_id` | строка | ID чата для алертов | +| `thresholds.cpu_percent` | число | Порог загрузки CPU, % | +| `thresholds.ram_avail_mb` | число | Минимум доступной RAM, МБ | +| `thresholds.disk_percent` | число | Порог занятости диска, % | +| `http_checks` | массив | `{name, url, expected}` | +| `tcp_checks` | массив | `{name, host, port}` | +| `netdata.base_url` | строка | URL Netdata | + +`expected` — список допустимых HTTP-кодов, например `[200]` или `[200, 401]`. + +## Формат алертов + +**Первый падение:** + +``` +⚠️ Server alert · 192.168.0.5 + +🔴 Gitea: HTTP 502 (expected 200), 12ms +``` + +**Напоминание:** + +``` +⏰ Still down · 192.168.0.5 + +🔴 Gitea: HTTP 502 (expected 200), 12ms +``` + +**Восстановление:** + +``` +✅ Recovered · 192.168.0.5 + +🟢 Gitea: HTTP 200, 10ms +``` + +**Алармы Netdata:** + +``` +⚠️ Server alert · 192.168.0.5 + +🔴 Netdata alarms: CPU_USAGE [CRITICAL]: 95 — ... +``` + +## Лог-режим без токена + +Если `TELEGRAM_BOT_TOKEN` пустой, сообщения не отправляются, а пишутся в лог `[TELEGRAM LOG-ONLY]`. Это удобно для отладки. + +## Локальный запуск без Docker + +```bash +# состояние пишется в ./state.json, а не в /data +STATE_FILE=./state.json python3 monitor.py --once +``` + +Флаг `--once` запускает один цикл проверок и завершает работу. Без `STATE_FILE` сервис попытается писать в `/data` (вне Docker это обычно недоступно — в этом случае состояние просто не сохранится, а первый запуск будет считаться baseline-прогоном). + +## Запуск тестов + +```bash +python3 -m unittest discover -s tests -v +``` + +## Troubleshooting + +**Контейнер не видит хост:** + +- Убедитесь, что у VPS есть маршрут к `192.168.0.5`. +- Проверьте firewall на Unraid/VPS. +- Для тестов можно запустить `docker run --rm --network host ...`. + +**Ping в контейнере:** + +- Внутри `python:3.12-alpine` может не быть прав на ICMP. +- `ping_check` деградирует gracefully: возвращает `ok=True` с сообщением `ping unavailable`, чтобы не ломать весь цикл. + +**Состояние не сохраняется:** + +- Проверьте, что volume `monitor-state` смонтирован в `/data`. +- В образе `/data` принадлежит пользователю `app` (uid 10001). diff --git a/SPEC.md b/SPEC.md new file mode 100644 index 0000000..50182b0 --- /dev/null +++ b/SPEC.md @@ -0,0 +1,127 @@ +# ServerMonitor — Dockerized server monitoring service + +Build a complete, production-quality server monitoring service in this repository. +The service runs inside a Docker container on a REMOTE VPS and monitors a home server (Unraid at 192.168.0.5) over the network. It periodically runs health checks and sends Telegram alerts on failures, recovery messages when services come back, and periodic reminders while a service stays down. + +## Hard constraints + +- Python 3.12, **stdlib ONLY** (urllib, socket, json, subprocess, time, signal, logging, threading, unittest). NO third-party dependencies, no pip installs, no requests/pyyaml. +- Must run on `python:3.12-alpine` (busybox, no bash — only `sh`). +- The main loop must NEVER crash because of a single failed check. Every check is wrapped in try/except, errors are logged, loop continues. +- Every network operation has a timeout. +- Graceful shutdown on SIGTERM/SIGINT: save state, log "shutting down", exit 0. +- Non-root user in the Docker image. +- Commit NOTHING and do not run git. Just write the files. + +## Files to create + +### 1. `checks.py` +Check functions, each returns a tuple `(ok: bool, message: str)` and never raises: + +- `ping_check(host, timeout)` — `subprocess.run(['ping', '-c', '1', '-W', str(timeout), host], ...)`. Busybox ping: `-W` is seconds. If ping is unavailable or permission denied (no ICMP in container), log a warning once and return ok=True with message "ping unavailable" (degrade gracefully — do not fail the check). +- `http_check(name, url, expected_codes, timeout)` — urllib.request with timeout; returns ok if HTTP status in expected_codes; message includes status, latency ms. Distinguish connection error / timeout / bad status. Follow no redirects. +- `tcp_check(name, host, port, timeout)` — socket.create_connection; ok if connects; message includes latency ms. +- `netdata_alarms(base_url, timeout)` — GET `{base_url}/api/v1/alarms?active` . Response JSON: `{"hostname":..., "status": true, "alarms": {name: {status, info, value, ...}}}`. If `status` is false → not ok. If `alarms` dict non-empty → not ok; message lists each alarm as `NAME [STATUS]: value — info`. If alarms empty → ok. `status: true` means Netdata itself is fine (it's `false` when Netdata cannot evaluate). +- `netdata_metrics(base_url, timeout)` — collect host metrics via Netdata data API to evaluate thresholds. Use `GET {base_url}/api/v1/data?chart=system.cpu&after=-1&points=1&format=json` (dimensions include `idle`, plus user/system/iowait/etc.; values are percentages; utilization = 100 - idle) and `GET {base_url}/api/v1/data?chart=system.ram&after=-1&points=1&format=json` (dims: used, free, cached, buffers in MB). Also `GET {base_url}/api/v1/data?chart=disk.space._&after=-1&points=1&format=json` — actually use `chart=disk.space` (it aggregates or errors; handle both, and if the chart is missing just skip disk) for per-mount used percent (dim name = mount path, values are percent used; dimension `avail` may exist — prefer the dims whose names are paths). Return dict with keys `cpu_percent`, `ram_used_mb`, `ram_total_mb` (from `system.ram` first value of `used`+`free`+`cached`+`buffers` OR parse `/proc/meminfo`-style — no, keep it to the API; ram_total_mb = sum of used/free/cached/buffers), `disk_used_percent` (max over mounts). Wrap EVERYTHING in try/except; on any error return None values and log a warning (metrics are best-effort, never alert on inability to fetch — only on exceeding thresholds when data IS available). +- Also `netdata_info(base_url, timeout)` — GET `{base_url}/api/v1/info`, ok if JSON with `version` present. (This is effectively covered by the HTTP check on :19999 root; you may skip if redundant.) + +### 2. `notify.py` +- `send_telegram(bot_token, chat_id, text, timeout)` — POST `https://api.telegram.org/bot/sendMessage` with form-encoded `chat_id`, `text`, `parse_mode=HTML`, `disable_web_page_preview=true`. Return bool. If bot_token empty → return True without sending (log-only mode) and log the would-be message at INFO. +- `escape_html(text)` — escape & < > for parse_mode=HTML. +- `class StateStore` — JSON file persistence (default path `/data/state.json`): dict keyed by check name: `{"state": "OK"|"CRIT", "last_alert_ts": float, "alert_count": int}`. Methods: `load()`, `save()`, `get(name)`, `set(name, state, last_alert_ts, alert_count)`. Atomic write (write temp file + os.replace). Never crash if file corrupt — start fresh. +- `class AlertEngine` — dedup logic: + - `should_alert(name, now, state_store, reminder_interval)` → (bool, is_recovery): + - CRIT and prev state != CRIT → alert (first failure). + - CRIT and prev state == CRIT and now - last_alert_ts >= reminder_interval → reminder (re-alert). + - OK and prev state == CRIT → recovery message. + - All alert decisions go through here; tests target this class. + +### 3. `monitor.py` +Main entry point: +- Load `config.json` from the same directory as the script (allow override via env `CONFIG_FILE`). +- Env overrides applied after config load: `HOST`, `CHECK_INTERVAL_SEC`, `TELEGRAM_BOT_TOKEN`, `TELEGRAM_CHAT_ID`, `STATE_FILE`. +- Setup logging: stdout, format `%(asctime)s %(levelname)s %(message)s`, INFO level. +- Register SIGTERM/SIGINT handler → set a threading.Event, main loop exits cleanly, state saved. +- Main loop (every `check_interval_sec`, default 60): + 1. Run all checks: ping, each http_check, each tcp_check, netdata_alarms, netdata_metrics thresholds (cpu_percent > threshold → CRIT "CPU 95%"; ram_avail_mb < threshold → CRIT; available = ram_total - ram_used; disk_used_percent > threshold → CRIT). + 2. For each check that is CRIT or recovered: decide alert via AlertEngine, send Telegram (sequential is fine, but keep total loop time bounded; timeouts are short). + 3. Log every check result at INFO: `Gitea OK (200, 15ms)` / `Gitea FAIL (502, expected [200], 12ms)`. + 4. On alert: log at WARNING. Message format (HTML): + - Failure: `⚠️ Server alert\n\n🔴 Gitea: HTTP 502 (expected 200), 12ms` (include host name from config `host` in header line: `Server alert · 192.168.0.5`). + - Reminder: `⏰ Still down · 192.168.0.5\n\n🔴 Gitea: HTTP 502 (expected 200), 12ms` + `\n(already 5 min)` — track downtime duration if easy; otherwise skip duration. + - Recovery: `✅ Recovered · 192.168.0.5\n\n🟢 Gitea: HTTP 200, 10ms`. + - Netdata alarms: `🔴 Netdata alarms: CPU_USAGE [CRITICAL]: 95 — ...` + 5. Save state each iteration. +- On first run with no state file: do NOT alert immediately for pre-existing failures — establish baseline: first iteration only logs and records state without sending alerts (prevents alert storm on deploy/restart). Implement via `state_file` missing + a `baseline_done` flag in state (or: treat first iteration as baseline when state file absent). +- `--once` CLI flag: run a single check cycle and exit (useful for cron/testing). + +### 4. `config.json` +Default config (services discovered on the target server): +```json +{ + "host": "192.168.0.5", + "check_interval_sec": 60, + "reminder_interval_sec": 1800, + "timeout_sec": 5, + "state_file": "/data/state.json", + "telegram": {"bot_token": "", "chat_id": ""}, + "thresholds": {"cpu_percent": 90, "ram_avail_mb": 2048, "disk_percent": 90}, + "ping": {"enabled": true}, + "http_checks": [ + {"name": "Gitea", "url": "http://192.168.0.5:3000/", "expected": [200]}, + {"name": "Unraid WebUI", "url": "http://192.168.0.5/", "expected": [200, 301, 302, 307]}, + {"name": "MetaCubeXD", "url": "http://192.168.0.5:9090/", "expected": [200]}, + {"name": "Netdata", "url": "http://192.168.0.5:19999/", "expected": [200]}, + {"name": "Plex", "url": "http://192.168.0.5:32400/", "expected": [200, 401]}, + {"name": "Nginx 8080", "url": "http://192.168.0.5:8080/", "expected": [200]} + ], + "tcp_checks": [ + {"name": "MySQL", "host": "192.168.0.5", "port": 3306}, + {"name": "SMB", "host": "192.168.0.5", "port": 445}, + {"name": "Proxy 1984", "host": "192.168.0.5", "port": 1984} + ], + "netdata": {"base_url": "http://192.168.0.5:19999", "check_alarms": true, "check_metrics": true} +} +``` + +### 5. `Dockerfile` +- `FROM python:3.12-alpine` +- create user `app` (adduser -D -u 10001 app), WORKDIR /app, copy files, `USER app` +- ensure `/data` exists and is writable by app (mkdir -p /data && chown) — note: /data will be a volume; make it owned by app in image. +- `ENV PYTHONUNBUFFERED=1` +- `CMD ["python", "-u", "/app/monitor.py"]` +- Healthcheck: use python one-liner checking state.json mtime freshness: `python -c "import os,time; p='/data/state.json'; raise SystemExit(0 if os.path.exists(p) and time.time()-os.path.getmtime(p)<300 else 1)"` — interval 60s, retries 3, start_period 30s. + +### 6. `docker-compose.yml` +- service `server-monitor`: build `.`, `restart: unless-stopped`, `env_file: .env`, volumes: `monitor-state:/data`, logging: `{driver: json-file, options: {max-size: "10m", max-file: "3"}}`. + +### 7. `.env.example` +``` +# Telegram bot for alerts (get from @BotFather). Leave empty for log-only mode. +TELEGRAM_BOT_TOKEN= +# Your chat id (get from @userinfobot) +TELEGRAM_CHAT_ID= +# Check interval in seconds (default 60) +CHECK_INTERVAL_SEC=60 +``` + +### 8. `README.md` — in Russian +Sections: что это; как это работает (схема: VPS → чеки → Telegram); быстрый старт (скопировать .env.example → .env, вписать токен, docker compose up -d --build); как получить токен бота (BotFather) и chat_id (@userinfobot); настройка чеков в config.json (таблица полей); формат алертов (примеры); лог-режим без токена; локальный запуск без Docker (python3 monitor.py --once); запуск тестов (python3 -m unittest discover -s tests -v); troubleshooting (контейнер не видит хост: сеть, firewall; ping в контейнере). + +### 9. `tests/test_checks.py` — unittest, stdlib only +- `http_check` returns ok for 200 expected 200: spin up `http.server` on 127.0.0.1:0 in a thread, hit it. +- `http_check` fails on wrong status (e.g. expected [200], server returns 404): use a handler that returns 404. +- `http_check` connection refused → not ok (closed port). +- `tcp_check` ok on listening socket; fail on closed port. +- `netdata_alarms` parsing: monkeypatch `urllib.request.urlopen` to return a fake response JSON (active alarms present → not ok, message contains alarm name; empty alarms → ok; status false → not ok). +- `AlertEngine` dedup: first CRIT alerts; second CRIT within reminder window does NOT alert; CRIT after reminder window alerts again; OK after CRIT sends recovery exactly once. +- `StateStore` round-trip and corrupt-file resilience (write garbage, load → empty dict). +- `escape_html` escapes `&<>`. +All tests must pass: `python3 -m unittest discover -s tests -v` + +## Quality bar +- Clean, readable, commented code (docstrings on functions). +- No print() in library modules — use logging. +- No global mutable state in checks.py (pass config around). +- File count: exactly the files above (SPEC.md already exists, don't modify it). +- Total code should be compact but complete — prefer clarity over cleverness. diff --git a/checks.py b/checks.py new file mode 100644 index 0000000..f779fe7 --- /dev/null +++ b/checks.py @@ -0,0 +1,330 @@ +"""Health check helpers for ServerMonitor. + +All check functions return a tuple ``(ok, message)`` and never raise. +""" + +import json +import logging +import socket +import subprocess +import time +import urllib.parse +import urllib.request +from urllib.error import HTTPError, URLError + +logger = logging.getLogger("servermonitor.checks") + +# Tracks whether the ping unavailability warning has been logged already. +_ping_unavailable_logged = False + + +def _safe_message(text, max_len=250): + """Return a single-line string truncated to ``max_len`` characters.""" + text = text.replace("\n", " ") + if len(text) > max_len: + text = text[: max_len - 1] + "…" + return text + + +def ping_check(host, timeout): + """Ping a host once and return ``(ok, message)``. + + On busybox (Alpine) ``-W`` is seconds. If ICMP is unavailable or disallowed + the check degrades gracefully: it returns ``ok=True`` with a message that + says ping is unavailable, so the service does not fail purely because the + container cannot ping. + """ + global _ping_unavailable_logged + + try: + start = time.monotonic() + result = subprocess.run( + ["ping", "-c", "1", "-W", str(timeout), host], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + timeout=timeout + 2, + ) + latency = int((time.monotonic() - start) * 1000) + + if result.returncode == 0: + return True, f"ping ok, {latency}ms" + + # Failure: host unreachable or no reply + return False, f"ping failed (code {result.returncode})" + + except FileNotFoundError: + if not _ping_unavailable_logged: + logger.warning("ping binary not available, degrading ping check") + _ping_unavailable_logged = True + return True, "ping unavailable" + + except PermissionError: + if not _ping_unavailable_logged: + logger.warning("ping permission denied, degrading ping check") + _ping_unavailable_logged = True + return True, "ping unavailable" + + except subprocess.TimeoutExpired: + return False, f"ping timeout after {timeout + 2}s" + + except Exception as exc: # noqa: BLE001 + return False, f"ping error: {exc}" + + +def http_check(name, url, expected_codes, timeout): + """Check that ``url`` returns a status code in ``expected_codes``. + + Follows no redirects. Returns ``(ok, message)`` where the message includes + the status code and latency. + """ + expected_codes = set(expected_codes) + start = time.monotonic() + + try: + req = urllib.request.Request( + url, + method="GET", + headers={"User-Agent": "ServerMonitor/1.0"}, + ) + + # Custom redirect handler that does not follow redirects, so the + # original status code (e.g. 301/302) is preserved. + class NoRedirectHandler(urllib.request.HTTPRedirectHandler): + def http_error_302(self, req, fp, code, msg, headers): + return fp + + http_error_301 = http_error_303 = http_error_307 = http_error_302 + + opener = urllib.request.build_opener(NoRedirectHandler) + with opener.open(req, timeout=timeout) as response: + status = response.status + latency = int((time.monotonic() - start) * 1000) + if status in expected_codes: + return True, f"HTTP {status}, {latency}ms" + return False, f"HTTP {status} (expected {sorted(expected_codes)}), {latency}ms" + + except HTTPError as exc: + latency = int((time.monotonic() - start) * 1000) + status = exc.code + if status in expected_codes: + return True, f"HTTP {status}, {latency}ms" + return False, f"HTTP {status} (expected {sorted(expected_codes)}), {latency}ms" + + except URLError as exc: + reason = str(exc.reason) + return False, f"connection error: {_safe_message(reason)}" + + except TimeoutError: + return False, f"timeout after {timeout}s" + + except Exception as exc: # noqa: BLE001 + return False, f"HTTP error: {_safe_message(str(exc))}" + + +def tcp_check(name, host, port, timeout): + """Check that a TCP connection to ``host:port`` succeeds. + + Returns ``(ok, message)`` with the latency in milliseconds. + """ + start = time.monotonic() + sock = None + try: + sock = socket.create_connection((host, port), timeout=timeout) + latency = int((time.monotonic() - start) * 1000) + return True, f"TCP ok, {latency}ms" + except socket.timeout: + return False, f"TCP timeout after {timeout}s" + except OSError as exc: + return False, f"TCP error: {_safe_message(str(exc))}" + except Exception as exc: # noqa: BLE001 + return False, f"TCP error: {_safe_message(str(exc))}" + finally: + if sock is not None: + try: + sock.close() + except Exception: # noqa: BLE001 + pass + + +def _netdata_json(url, timeout): + """Fetch and parse a Netdata JSON endpoint. Returns dict or None.""" + req = urllib.request.Request(url, headers={"User-Agent": "ServerMonitor/1.0"}) + with urllib.request.urlopen(req, timeout=timeout) as response: + data = response.read() + return json.loads(data.decode("utf-8")) + + +def netdata_alarms(base_url, timeout): + """Check Netdata active alarms. + + ``status: false`` in the response or any non-empty alarms object means the + check fails. Returns ``(ok, message)``. + """ + url = f"{base_url}/api/v1/alarms?active" + try: + payload = _netdata_json(url, timeout) + except Exception as exc: # noqa: BLE001 + return False, f"Netdata alarms API error: {_safe_message(str(exc))}" + + if not isinstance(payload, dict): + return False, "Netdata alarms invalid response format" + + net_status = payload.get("status", True) + alarms = payload.get("alarms", {}) + + if not net_status: + return False, "Netdata status false" + + if not alarms: + return True, "Netdata alarms empty" + + parts = [] + for name, details in alarms.items(): + if not isinstance(details, dict): + details = {} + status = details.get("status", "?") + value = details.get("value", "?") + info = details.get("info", "") + parts.append(f"{name} [{status}]: {value} — {info}") + + return False, "Netdata alarms: " + "; ".join(parts) + + +def _row_values(labels, row): + """Convert a Netdata data row into a ``{label: float or None}`` dict. + + The ``time`` dimension is skipped. Missing values (``null``) become + ``None``. + """ + values = {} + for i, label in enumerate(labels): + if label == "time": + continue + try: + raw = row[i] + values[label] = float(raw) if raw is not None else None + except (TypeError, ValueError, IndexError): + values[label] = None + return values + + +def netdata_metrics(base_url, timeout): + """Collect Netdata metrics for threshold checks. + + Returns a dict with keys ``cpu_percent``, ``ram_used_mb``, ``ram_total_mb``, + ``disk_used_percent``. Missing values are represented as ``None``. + """ + result = { + "cpu_percent": None, + "ram_used_mb": None, + "ram_total_mb": None, + "disk_used_percent": None, + } + + try: + cpu_url = ( + f"{base_url}/api/v1/data?chart=system.cpu&after=-1&points=1&format=json" + ) + cpu_data = _netdata_json(cpu_url, timeout) + if isinstance(cpu_data, dict): + labels = cpu_data.get("labels", []) + data_rows = cpu_data.get("data", []) + if data_rows and labels: + values = _row_values(labels, data_rows[0]) + if values.get("idle") is not None: + # Older Netdata: idle dimension present. + result["cpu_percent"] = max(0.0, 100.0 - values["idle"]) + else: + # Netdata v2: only busy dimensions, their sum is ~utilization. + total = sum(v for v in values.values() if v is not None) + result["cpu_percent"] = max(0.0, min(100.0, total)) + except Exception as exc: # noqa: BLE001 + logger.warning("Netdata CPU metrics fetch failed: %s", exc) + + try: + ram_url = ( + f"{base_url}/api/v1/data?chart=system.ram&after=-1&points=1&format=json" + ) + ram_data = _netdata_json(ram_url, timeout) + if isinstance(ram_data, dict): + labels = ram_data.get("labels", []) + data_rows = ram_data.get("data", []) + if data_rows and labels: + values = _row_values(labels, data_rows[0]) + total = 0.0 + for label in ("used", "free", "cached", "buffers"): + value = values.get(label) + if value is not None: + total += value + result["ram_used_mb"] = values.get("used") + result["ram_total_mb"] = total if total > 0 else None + except Exception as exc: # noqa: BLE001 + logger.warning("Netdata RAM metrics fetch failed: %s", exc) + + try: + # Discover per-mount disk charts (named disk_space. on Netdata v2) + # and take the maximum used-percent across them. + charts_data = _netdata_json(f"{base_url}/api/v1/charts", timeout) + disk_charts = [] + if isinstance(charts_data, dict): + disk_charts = [ + key + for key in charts_data.get("charts", {}) + if key.startswith("disk_space.") + ][:8] + + max_used = None + for chart in disk_charts: + try: + disk_url = ( + f"{base_url}/api/v1/data" + f"?chart={urllib.parse.quote(chart)}&after=-1&points=1&format=json" + ) + disk_data = _netdata_json(disk_url, timeout) + if not isinstance(disk_data, dict): + continue + labels = disk_data.get("labels", []) + data_rows = disk_data.get("data", []) + if not data_rows or not labels: + continue + values = _row_values(labels, data_rows[0]) + used = values.get("used") + avail = values.get("avail") + if used is not None and avail is not None: + # Charts may report absolute values (GB/MB) or percents; + # compute the used-percent ourselves so units do not matter. + reserved = values.get("reserved for root") or 0.0 + denominator = used + avail + reserved + candidate = (used / denominator * 100.0) if denominator > 0 else None + else: + candidates = [ + v + for k, v in values.items() + if v is not None and k != "avail" + ] + candidate = max(candidates) if candidates else None + if candidate is not None: + max_used = ( + candidate if max_used is None else max(max_used, candidate) + ) + except Exception as exc: # noqa: BLE001 + logger.warning("Netdata disk chart %s fetch failed: %s", chart, exc) + + result["disk_used_percent"] = max_used + except Exception as exc: # noqa: BLE001 + logger.warning("Netdata disk metrics fetch failed: %s", exc) + + return result + + +def netdata_info(base_url, timeout): + """Check that the Netdata info endpoint is reachable and contains version.""" + url = f"{base_url}/api/v1/info" + try: + payload = _netdata_json(url, timeout) + if isinstance(payload, dict) and "version" in payload: + return True, f"Netdata {payload.get('version')}" + return False, "Netdata info missing version" + except Exception as exc: # noqa: BLE001 + return False, f"Netdata info error: {_safe_message(str(exc))}" diff --git a/config.json b/config.json new file mode 100644 index 0000000..f35d118 --- /dev/null +++ b/config.json @@ -0,0 +1,24 @@ +{ + "host": "192.168.0.5", + "check_interval_sec": 60, + "reminder_interval_sec": 1800, + "timeout_sec": 5, + "state_file": "/data/state.json", + "telegram": {"bot_token": "", "chat_id": ""}, + "thresholds": {"cpu_percent": 90, "ram_avail_mb": 2048, "disk_percent": 90}, + "ping": {"enabled": true}, + "http_checks": [ + {"name": "Gitea", "url": "http://192.168.0.5:3000/", "expected": [200]}, + {"name": "Unraid WebUI", "url": "http://192.168.0.5/", "expected": [200, 301, 302, 307]}, + {"name": "MetaCubeXD", "url": "http://192.168.0.5:9090/ui/", "expected": [200]}, + {"name": "Netdata", "url": "http://192.168.0.5:19999/", "expected": [200]}, + {"name": "Plex", "url": "http://192.168.0.5:32400/", "expected": [200, 401]}, + {"name": "Nginx 8080", "url": "http://192.168.0.5:8080/", "expected": [200]} + ], + "tcp_checks": [ + {"name": "MySQL", "host": "192.168.0.5", "port": 3306}, + {"name": "SMB", "host": "192.168.0.5", "port": 445}, + {"name": "Proxy 1984", "host": "192.168.0.5", "port": 1984} + ], + "netdata": {"base_url": "http://192.168.0.5:19999", "check_alarms": true, "check_metrics": true} +} diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..92d2f1c --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,17 @@ +version: "3.8" + +services: + server-monitor: + build: . + restart: unless-stopped + env_file: .env + volumes: + - monitor-state:/data + logging: + driver: json-file + options: + max-size: "10m" + max-file: "3" + +volumes: + monitor-state: diff --git a/monitor.py b/monitor.py new file mode 100644 index 0000000..c80e8db --- /dev/null +++ b/monitor.py @@ -0,0 +1,296 @@ +"""ServerMonitor main loop. + +Runs periodic health checks and sends Telegram alerts on failures, reminders +while services stay down, and recovery messages when they come back. +""" + +import argparse +import json +import logging +import os +import signal +import sys +import threading +import time +from urllib.parse import urlparse, urlunparse + +from checks import http_check, netdata_alarms, netdata_metrics, ping_check, tcp_check +from notify import AlertEngine, StateStore, escape_html, send_telegram + +logger = logging.getLogger("servermonitor.monitor") + +shutdown_event = threading.Event() + + +def load_config(path): + """Load the JSON configuration file.""" + with open(path, "r", encoding="utf-8") as f: + return json.load(f) + + +def _replace_host_in_url(url, new_host): + """Replace the netloc host in a URL without changing the port or path.""" + parsed = urlparse(url) + # Preserve user:pass and port; replace only the hostname. + netloc = new_host + if parsed.port is not None: + netloc = f"{new_host}:{parsed.port}" + if parsed.username is not None: + prefix = f"{parsed.username}" + if parsed.password is not None: + prefix += f":{parsed.password}" + netloc = f"{prefix}@{netloc}" + return urlunparse(parsed._replace(netloc=netloc)) + + +def apply_env_overrides(config): + """Override config values from environment variables.""" + new_host = None + if "HOST" in os.environ: + new_host = os.environ["HOST"] + config["host"] = new_host + + if "CHECK_INTERVAL_SEC" in os.environ: + try: + config["check_interval_sec"] = int(os.environ["CHECK_INTERVAL_SEC"]) + except ValueError: + logger.warning("Invalid CHECK_INTERVAL_SEC, using config value") + if "TELEGRAM_BOT_TOKEN" in os.environ: + config["telegram"]["bot_token"] = os.environ["TELEGRAM_BOT_TOKEN"] + if "TELEGRAM_CHAT_ID" in os.environ: + config["telegram"]["chat_id"] = os.environ["TELEGRAM_CHAT_ID"] + if "STATE_FILE" in os.environ: + config["state_file"] = os.environ["STATE_FILE"] + + if new_host: + for check in config.get("http_checks", []): + check["url"] = _replace_host_in_url(check.get("url", ""), new_host) + for check in config.get("tcp_checks", []): + check["host"] = new_host + netdata_cfg = config.get("netdata", {}) + if netdata_cfg.get("base_url"): + netdata_cfg["base_url"] = _replace_host_in_url(netdata_cfg["base_url"], new_host) + + +def setup_logging(): + """Configure logging to stdout.""" + logging.basicConfig( + stream=sys.stdout, + format="%(asctime)s %(levelname)s %(message)s", + level=logging.INFO, + ) + + +def shutdown_handler(signum, frame): + """Handle SIGTERM/SIGINT gracefully.""" + logger.info("shutting down") + shutdown_event.set() + + +def run_checks(config): + """Run every configured check and return a list of results. + + Each result is a tuple ``(name, ok, message)``. + """ + host = config["host"] + timeout = config.get("timeout_sec", 5) + results = [] + + if config.get("ping", {}).get("enabled", True): + ok, msg = ping_check(host, timeout) + results.append(("ping", ok, msg)) + + for check in config.get("http_checks", []): + ok, msg = http_check( + check["name"], + check["url"], + check.get("expected", [200]), + timeout, + ) + results.append((check["name"], ok, msg)) + + for check in config.get("tcp_checks", []): + ok, msg = tcp_check( + check["name"], + check.get("host", host), + check["port"], + timeout, + ) + results.append((check["name"], ok, msg)) + + netdata_cfg = config.get("netdata", {}) + base_url = netdata_cfg.get("base_url", f"http://{host}:19999") + + if netdata_cfg.get("check_alarms", True): + ok, msg = netdata_alarms(base_url, timeout) + results.append(("netdata_alarms", ok, msg)) + + if netdata_cfg.get("check_metrics", True): + metrics = netdata_metrics(base_url, timeout) + thresholds = config.get("thresholds", {}) + + cpu = metrics.get("cpu_percent") + cpu_threshold = thresholds.get("cpu_percent", 90) + if cpu is not None and cpu > cpu_threshold: + results.append(( + "CPU", + False, + f"CPU {cpu:.1f}% (threshold {cpu_threshold}%)", + )) + elif cpu is not None: + results.append(("CPU", True, f"CPU {cpu:.1f}%")) + + ram_used = metrics.get("ram_used_mb") + ram_total = metrics.get("ram_total_mb") + ram_threshold = thresholds.get("ram_avail_mb", 2048) + if ram_used is not None and ram_total is not None: + ram_avail = ram_total - ram_used + if ram_avail < ram_threshold: + results.append(( + "RAM", + False, + f"RAM available {ram_avail:.0f}MB (threshold {ram_threshold}MB)", + )) + else: + results.append(("RAM", True, f"RAM available {ram_avail:.0f}MB")) + + disk = metrics.get("disk_used_percent") + disk_threshold = thresholds.get("disk_percent", 90) + if disk is not None and disk > disk_threshold: + results.append(( + "Disk", + False, + f"Disk {disk:.1f}% used (threshold {disk_threshold}%)", + )) + elif disk is not None: + results.append(("Disk", True, f"Disk {disk:.1f}% used")) + + return results + + +def format_message(host, name, ok, message, is_recovery, alert_count): + """Format a Telegram HTML alert message.""" + safe_name = escape_html(name) + safe_message = escape_html(message) + safe_host = escape_html(host) + + if is_recovery: + header = f"✅ Recovered · {safe_host}" + body = f"🟢 {safe_name}: {safe_message}" + return f"{header}\n\n{body}" + + if ok: + # Should not normally alert on OK, but handle it gracefully. + header = f"✅ Recovered · {safe_host}" + body = f"🟢 {safe_name}: {safe_message}" + return f"{header}\n\n{body}" + + if name == "netdata_alarms": + header = f"⚠️ Server alert · {safe_host}" + body = f"🔴 Netdata alarms: {safe_message}" + return f"{header}\n\n{body}" + + if alert_count > 1: + header = f"⏰ Still down · {safe_host}" + body = f"🔴 {safe_name}: {safe_message}" + return f"{header}\n\n{body}" + + header = f"⚠️ Server alert · {safe_host}" + body = f"🔴 {safe_name}: {safe_message}" + return f"{header}\n\n{body}" + + +def main_loop(config, state_store, alert_engine, once=False): + """Run the monitoring loop until shutdown or a single cycle if ``once``.""" + reminder_interval = config.get("reminder_interval_sec", 1800) + host = config["host"] + bot_token = config["telegram"].get("bot_token", "") + chat_id = config["telegram"].get("chat_id", "") + timeout = config.get("timeout_sec", 5) + + baseline = not os.path.exists(state_store.path) + + while not shutdown_event.is_set(): + now = time.time() + results = run_checks(config) + + for name, ok, message in results: + logger.info("%s %s (%s)", name, "OK" if ok else "FAIL", message) + + current_state = "OK" if ok else "CRIT" + alert_engine.set_state(name, current_state) + should_send, is_recovery = alert_engine.should_alert( + name, now, state_store, reminder_interval + ) + + if baseline: + # First run with no state file: establish baseline, record state, + # do not alert. alert_count stays 0 so the next real failure is + # treated as the first alert. + state_store.set(name, current_state, 0.0, 0) + continue + + if should_send: + state_entry = state_store.get(name) + alert_count = state_entry["alert_count"] + last_alert_ts = state_entry["last_alert_ts"] + text = format_message( + host, + name, + ok, + message, + is_recovery, + alert_count, + ) + logger.warning("Alert: %s", text.replace("\n", " ")) + send_telegram(bot_token, chat_id, text, timeout) + + baseline = False + state_store.save() + + if once: + break + + try: + shutdown_event.wait(config.get("check_interval_sec", 60)) + except Exception: # noqa: BLE001 + break + + +def main(): + parser = argparse.ArgumentParser(description="ServerMonitor") + parser.add_argument("--once", action="store_true", help="Run one cycle and exit") + args = parser.parse_args() + + setup_logging() + + script_dir = os.path.dirname(os.path.abspath(__file__)) + config_path = os.environ.get("CONFIG_FILE", os.path.join(script_dir, "config.json")) + + try: + config = load_config(config_path) + except Exception as exc: # noqa: BLE001 + logger.error("Failed to load config from %s: %s", config_path, exc) + sys.exit(1) + + apply_env_overrides(config) + + state_store = StateStore(config.get("state_file", "/data/state.json")) + state_store.load() + + alert_engine = AlertEngine() + + signal.signal(signal.SIGTERM, shutdown_handler) + signal.signal(signal.SIGINT, shutdown_handler) + + try: + main_loop(config, state_store, alert_engine, once=args.once) + except Exception as exc: # noqa: BLE001 + logger.exception("Unexpected error in main loop: %s", exc) + finally: + state_store.save() + logger.info("shutting down") + + +if __name__ == "__main__": + main() diff --git a/notify.py b/notify.py new file mode 100644 index 0000000..4601a30 --- /dev/null +++ b/notify.py @@ -0,0 +1,173 @@ +"""Notification, persistence, and alert deduplication logic.""" + +import json +import logging +import os +import tempfile +import time +import urllib.parse +import urllib.request + +logger = logging.getLogger("servermonitor.notify") + + +def escape_html(text): + """Escape ``&``, ``<`` and ``>`` for Telegram ``parse_mode=HTML``.""" + return text.replace("&", "&").replace("<", "<").replace(">", ">") + + +def send_telegram(bot_token, chat_id, text, timeout): + """Send an HTML message via Telegram. + + If ``bot_token`` is empty the function runs in log-only mode: it logs the + would-be message and returns ``True``. + """ + if not bot_token: + logger.info("[TELEGRAM LOG-ONLY] %s", text) + return True + + url = f"https://api.telegram.org/bot{bot_token}/sendMessage" + data = urllib.parse.urlencode( + { + "chat_id": chat_id, + "text": text, + "parse_mode": "HTML", + "disable_web_page_preview": "true", + } + ).encode("utf-8") + + req = urllib.request.Request( + url, + data=data, + method="POST", + headers={ + "Content-Type": "application/x-www-form-urlencoded", + "User-Agent": "ServerMonitor/1.0", + }, + ) + + try: + with urllib.request.urlopen(req, timeout=timeout) as response: + response.read() + return True + except Exception as exc: # noqa: BLE001 + logger.error("Failed to send Telegram message: %s", exc) + return False + + +class StateStore: + """JSON file persistence for check/alert state. + + The file is written atomically: a temporary file is created in the same + directory and then moved into place. If the existing file is corrupt, a + fresh empty state is returned. + """ + + def __init__(self, path="/data/state.json"): + self.path = path + self._state = {} + + def load(self): + """Load state from disk.""" + if not os.path.exists(self.path): + self._state = {} + return + + try: + with open(self.path, "r", encoding="utf-8") as f: + self._state = json.load(f) + if not isinstance(self._state, dict): + self._state = {} + except Exception as exc: # noqa: BLE001 + logger.warning("State file %s corrupt or unreadable: %s. Starting fresh.", self.path, exc) + self._state = {} + + def save(self): + """Persist state to disk atomically.""" + directory = os.path.dirname(os.path.abspath(self.path)) or "." + try: + os.makedirs(directory, exist_ok=True) + except Exception as exc: # noqa: BLE001 + logger.warning("Could not create state directory %s: %s", directory, exc) + return + + try: + fd, tmp_path = tempfile.mkstemp( + dir=directory, prefix="state", suffix=".json.tmp" + ) + with os.fdopen(fd, "w", encoding="utf-8") as f: + json.dump(self._state, f, indent=2) + f.write("\n") + os.replace(tmp_path, self.path) + except Exception as exc: # noqa: BLE001 + logger.error("Could not save state file %s: %s", self.path, exc) + + def get(self, name): + """Return the stored state for ``name`` or a default OK state.""" + entry = self._state.get(name) + if not isinstance(entry, dict): + return {"state": "OK", "last_alert_ts": 0.0, "alert_count": 0} + return { + "state": entry.get("state", "OK"), + "last_alert_ts": float(entry.get("last_alert_ts", 0.0)), + "alert_count": int(entry.get("alert_count", 0)), + } + + def set(self, name, state, last_alert_ts, alert_count): + """Set the stored state for ``name``.""" + self._state[name] = { + "state": state, + "last_alert_ts": float(last_alert_ts), + "alert_count": int(alert_count), + } + + +class AlertEngine: + """Decide whether a check should trigger a Telegram alert. + + - First CRIT transition: alert. + - Repeated CRIT inside the reminder window: suppress. + - Repeated CRIT after reminder window: reminder alert. + - OK transition after CRIT: recovery alert (exactly once). + + The caller should set the current observed state with ``set_state`` before + calling ``should_alert``. + """ + + def __init__(self): + self._current_states = {} + + def set_state(self, name, state): + """Set the freshly observed state for ``name`` (``OK`` or ``CRIT``).""" + self._current_states[name] = state + + def should_alert(self, name, now, state_store, reminder_interval): + """Return ``(should_send, is_recovery)``. + + ``state_store`` is updated in place when an alert decision is made. + """ + current_state = self._current_states.get(name, "OK") + prev = state_store.get(name) + prev_state = prev["state"] + last_alert_ts = prev["last_alert_ts"] + alert_count = prev["alert_count"] + + if current_state == "CRIT": + if prev_state != "CRIT": + state_store.set(name, "CRIT", now, alert_count + 1) + return True, False + if now - last_alert_ts >= reminder_interval: + state_store.set(name, "CRIT", now, alert_count + 1) + return True, False + return False, False + + if current_state == "OK" and prev_state == "CRIT": + state_store.set(name, "OK", last_alert_ts, alert_count) + # Only send a recovery message if an alert was actually sent for + # this check before (e.g. not after a baseline run that observed + # the failure without alerting). + if alert_count > 0: + return True, True + return False, False + + return False, False diff --git a/tests/test_checks.py b/tests/test_checks.py new file mode 100644 index 0000000..e96d346 --- /dev/null +++ b/tests/test_checks.py @@ -0,0 +1,353 @@ +"""Unit tests for ServerMonitor checks and notification logic.""" + +import http.server +import json +import os +import socket +import tempfile +import threading +import unittest +from io import BytesIO +from unittest.mock import patch + +from checks import http_check, netdata_alarms, netdata_metrics, tcp_check +from notify import AlertEngine, StateStore, escape_html + + +class TestHttpCheck(unittest.TestCase): + """Tests for ``http_check``.""" + + def _serve(self, handler_class, port=0): + """Start a threaded HTTP server and return its URL.""" + server = http.server.HTTPServer(("127.0.0.1", port), handler_class) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + + def cleanup(): + server.shutdown() + server.server_close() + + self.addCleanup(cleanup) + host, port = server.server_address + return f"http://{host}:{port}/" + + def test_ok_on_200(self): + """http_check returns ok when the status is in expected_codes.""" + url = self._serve(http.server.SimpleHTTPRequestHandler) + ok, message = http_check("test", url, [200], 5) + self.assertTrue(ok) + self.assertIn("HTTP 200", message) + + def test_fail_on_wrong_status(self): + """http_check fails when the server returns an unexpected status.""" + class NotFoundHandler(http.server.BaseHTTPRequestHandler): + def do_GET(self): + self.send_response(404) + self.end_headers() + + def log_message(self, fmt, *args): + pass + + url = self._serve(NotFoundHandler) + ok, message = http_check("test", url, [200], 5) + self.assertFalse(ok) + self.assertIn("HTTP 404", message) + self.assertIn("expected [200]", message) + + def test_fail_on_connection_refused(self): + """http_check reports failure when the port is closed.""" + # Find a closed port by binding to 0 and then immediately closing it. + sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + sock.bind(("127.0.0.1", 0)) + _, port = sock.getsockname() + sock.close() + url = f"http://127.0.0.1:{port}/" + ok, message = http_check("test", url, [200], 2) + self.assertFalse(ok) + self.assertIn("connection error", message) + + +class TestTcpCheck(unittest.TestCase): + """Tests for ``tcp_check``.""" + + def test_ok_on_listening_socket(self): + """tcp_check succeeds when something is listening.""" + sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + sock.bind(("127.0.0.1", 0)) + sock.listen(1) + _, port = sock.getsockname() + try: + ok, message = tcp_check("test", "127.0.0.1", port, 2) + self.assertTrue(ok) + self.assertIn("TCP ok", message) + finally: + sock.close() + + def test_fail_on_closed_port(self): + """tcp_check fails when the port is closed.""" + sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + sock.bind(("127.0.0.1", 0)) + _, port = sock.getsockname() + sock.close() + + ok, message = tcp_check("test", "127.0.0.1", port, 2) + self.assertFalse(ok) + self.assertIn("TCP error", message) + + +class FakeResponse: + """Minimal urllib response stand-in for monkeypatching.""" + + def __init__(self, payload_dict): + self._body = json.dumps(payload_dict).encode("utf-8") + + def read(self): + return self._body + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + +class TestNetdataAlarms(unittest.TestCase): + """Tests for ``netdata_alarms``.""" + + def test_active_alarms_present(self): + """netdata_alarms is not ok and names the alarm when alarms exist.""" + payload = { + "hostname": "unraid", + "status": True, + "alarms": { + "CPU_USAGE": { + "status": "CRITICAL", + "info": "CPU is high", + "value": 95, + } + }, + } + with patch("checks.urllib.request.urlopen", return_value=FakeResponse(payload)): + ok, message = netdata_alarms("http://netdata.test", 5) + self.assertFalse(ok) + self.assertIn("CPU_USAGE", message) + self.assertIn("CRITICAL", message) + self.assertIn("95", message) + + def test_empty_alarms_ok(self): + """netdata_alarms is ok when alarms are empty.""" + payload = { + "hostname": "unraid", + "status": True, + "alarms": {}, + } + with patch("checks.urllib.request.urlopen", return_value=FakeResponse(payload)): + ok, message = netdata_alarms("http://netdata.test", 5) + self.assertTrue(ok) + self.assertIn("empty", message) + + def test_status_false_not_ok(self): + """netdata_alarms is not ok when the top-level status is false.""" + payload = { + "hostname": "unraid", + "status": False, + "alarms": {}, + } + with patch("checks.urllib.request.urlopen", return_value=FakeResponse(payload)): + ok, message = netdata_alarms("http://netdata.test", 5) + self.assertFalse(ok) + self.assertIn("status false", message) + + +class TestNetdataMetrics(unittest.TestCase): + """Tests for ``netdata_metrics`` parsing (Netdata v2 shapes).""" + + def test_cpu_without_idle_uses_busy_sum(self): + """Netdata v2 has no idle dim: utilization is the sum of busy dims.""" + cpu_response = FakeResponse( + { + "labels": ["time", "user", "system", "iowait"], + "data": [[1785957961, 50.0, 30.0, 20.0]], + } + ) + with patch( + "checks.urllib.request.urlopen", return_value=cpu_response + ) as mocked: + metrics = netdata_metrics("http://netdata.test", 5) + self.assertEqual(metrics["cpu_percent"], 100.0) + + def test_cpu_with_idle(self): + """Older Netdata with idle dim: utilization = 100 - idle.""" + cpu_response = FakeResponse( + { + "labels": ["time", "idle", "user"], + "data": [[1785957961, 25.0, 75.0]], + } + ) + with patch( + "checks.urllib.request.urlopen", return_value=cpu_response + ) as mocked: + metrics = netdata_metrics("http://netdata.test", 5) + self.assertAlmostEqual(metrics["cpu_percent"], 75.0) + + def test_disk_discovery_and_max_used(self): + """disk_space.* charts are discovered and max used-percent is taken.""" + responses = [ + FakeResponse({"labels": ["time", "user"], "data": [[1, 10.0]]}), # cpu + FakeResponse( + { + "labels": ["time", "used", "free"], + "data": [[1, 2000.0, 1000.0]], + } + ), # ram + FakeResponse( + {"charts": {"disk_space./": {}, "disk_space./var": {}}} + ), # charts + FakeResponse( + {"labels": ["time", "avail", "used"], "data": [[1, 50.0, 45.3]]} + ), # disk / -> 45.3/(50+45.3) = 47.5% + FakeResponse( + {"labels": ["time", "avail", "used"], "data": [[1, 10.0, 88.0]]} + ), # disk /var -> 88/(10+88) = 89.8% + ] + with patch( + "checks.urllib.request.urlopen", + side_effect=responses, + ): + metrics = netdata_metrics("http://netdata.test", 5) + self.assertAlmostEqual(metrics["disk_used_percent"], 89.8, delta=0.01) + + def test_ram_totals(self): + """ram_used_mb and ram_total_mb are parsed from the ram chart.""" + responses = [ + FakeResponse({"labels": ["time", "user"], "data": [[1, 10.0]]}), # cpu + FakeResponse( + { + "labels": ["time", "free", "used", "cached", "buffers"], + "data": [[1, 1000.0, 2000.0, 3000.0, 100.0]], + } + ), # ram + FakeResponse({"charts": {}}), # no disk charts + ] + with patch( + "checks.urllib.request.urlopen", + side_effect=responses, + ): + metrics = netdata_metrics("http://netdata.test", 5) + self.assertEqual(metrics["ram_used_mb"], 2000.0) + self.assertEqual(metrics["ram_total_mb"], 6100.0) + + +class TestAlertEngine(unittest.TestCase): + """Tests for ``AlertEngine`` deduplication logic.""" + + def test_first_crit_alerts(self): + """The first CRIT transition triggers an alert.""" + store = StateStore() + store.load() + engine = AlertEngine() + engine.set_state("svc", "CRIT") + should, recovery = engine.should_alert("svc", 1000, store, 600) + self.assertTrue(should) + self.assertFalse(recovery) + + def test_second_crit_within_window_does_not_alert(self): + """A repeated CRIT inside the reminder window is suppressed.""" + store = StateStore() + store.load() + engine = AlertEngine() + engine.set_state("svc", "CRIT") + engine.should_alert("svc", 1000, store, 600) + engine.set_state("svc", "CRIT") + should, recovery = engine.should_alert("svc", 1200, store, 600) + self.assertFalse(should) + self.assertFalse(recovery) + + def test_crit_after_window_alerts_again(self): + """A CRIT after the reminder window triggers a reminder.""" + store = StateStore() + store.load() + engine = AlertEngine() + engine.set_state("svc", "CRIT") + engine.should_alert("svc", 1000, store, 600) + engine.set_state("svc", "CRIT") + should, recovery = engine.should_alert("svc", 1700, store, 600) + self.assertTrue(should) + self.assertFalse(recovery) + + def test_recovery_sent_once(self): + """A CRIT → OK transition triggers exactly one recovery message.""" + store = StateStore() + store.load() + engine = AlertEngine() + engine.set_state("svc", "CRIT") + engine.should_alert("svc", 1000, store, 600) + engine.set_state("svc", "OK") + should, recovery = engine.should_alert("svc", 1100, store, 600) + self.assertTrue(should) + self.assertTrue(recovery) + # Repeated OK must not alert again. + engine.set_state("svc", "OK") + should, recovery = engine.should_alert("svc", 1200, store, 600) + self.assertFalse(should) + self.assertFalse(recovery) + + + def test_no_recovery_without_prior_alert(self): + """A CRIT observed at baseline (never alerted) must not send recovery.""" + store = StateStore() + store.load() + # Simulate baseline: CRIT observed but alert_count stays 0. + store.set("svc", "CRIT", 0.0, 0) + engine = AlertEngine() + engine.set_state("svc", "OK") + should, recovery = engine.should_alert("svc", 1100, store, 600) + self.assertFalse(should) + self.assertFalse(recovery) + # State must still be updated to OK. + self.assertEqual(store.get("svc")["state"], "OK") + + +class TestStateStore(unittest.TestCase): + """Tests for ``StateStore`` persistence.""" + + def test_round_trip(self): + """State written to disk can be loaded back.""" + with tempfile.NamedTemporaryFile(delete=False) as tmp: + path = tmp.name + try: + store = StateStore(path) + store.load() + store.set("svc", "CRIT", 123.4, 7) + store.save() + + store2 = StateStore(path) + store2.load() + self.assertEqual(store2.get("svc"), {"state": "CRIT", "last_alert_ts": 123.4, "alert_count": 7}) + finally: + os.unlink(path) + + def test_corrupt_file_starts_fresh(self): + """A corrupt state file is handled gracefully and yields an empty state.""" + with tempfile.NamedTemporaryFile(delete=False, mode="w", encoding="utf-8") as tmp: + tmp.write("not valid json {{}}") + path = tmp.name + try: + store = StateStore(path) + store.load() + self.assertEqual(store.get("svc"), {"state": "OK", "last_alert_ts": 0.0, "alert_count": 0}) + finally: + os.unlink(path) + + +class TestEscapeHtml(unittest.TestCase): + """Tests for ``escape_html``.""" + + def test_escapes_special_chars(self): + """escape_html escapes &, < and >.""" + self.assertEqual(escape_html("a & b < c > d"), "a & b < c > d") + + +if __name__ == "__main__": + unittest.main()