commit c756195530734de143c842f8402a045b4fbfbae3 Author: MrYoshii Date: Sun Apr 12 04:15:02 2026 +0200 Initial commit: server-monitor v1 (Python) Kompletter Neustart als Python-Projekt. Portiert von zfs-monitor Bash v3, saubere Architektur: - config.py: TOML-Config mit dataclasses - state.py: zentrales State-Management (JSON statt Einzeldateien) - notify.py: Telegram + Persistent Alerting als wiederverwendbare Klasse - checks/zfs.py: ZFS Pool-Health, I/O-Fehler, Kapazität - checks/smart.py: SATA + NVMe SMART-Monitoring - checks/nut.py: NUT/USV-Monitoring (APC Back-UPS RS 1500G) diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..ff9a835 --- /dev/null +++ b/.gitignore @@ -0,0 +1,5 @@ +__pycache__/ +*.pyc +*.pyo +.env +*.toml.local diff --git a/checks/__init__.py b/checks/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/checks/base.py b/checks/base.py new file mode 100644 index 0000000..68c5f50 --- /dev/null +++ b/checks/base.py @@ -0,0 +1,38 @@ +from __future__ import annotations + +import logging +import subprocess +from abc import ABC, abstractmethod + +from config import Config +from notify import Alerter +from state import State + + +class Check(ABC): + def __init__(self, config: Config, state: State, alerter: Alerter) -> None: + self.config = config + self.state = state + self.alerter = alerter + self.log = logging.getLogger(self.__class__.__name__) + + @abstractmethod + def run(self) -> None: + pass + + def _run(self, cmd: list[str], timeout: int) -> str | None: + """Run command, return stdout (even on non-zero exit) or None on error.""" + try: + result = subprocess.run( + cmd, + capture_output=True, + text=True, + timeout=timeout, + ) + return result.stdout + except subprocess.TimeoutExpired: + self.log.warning("Timeout: %s", " ".join(cmd)) + return None + except (FileNotFoundError, OSError) as e: + self.log.warning("Fehler bei '%s': %s", cmd[0], e) + return None diff --git a/checks/nut.py b/checks/nut.py new file mode 100644 index 0000000..997143b --- /dev/null +++ b/checks/nut.py @@ -0,0 +1,169 @@ +from __future__ import annotations + +from checks.base import Check +from state import make_key + + +class NutCheck(Check): + def run(self) -> None: + for ups in self.config.nut.ups: + self.log.info("--- USV: %s ---", ups) + self._check_ups(ups) + + def _check_ups(self, ups_name: str) -> None: + key_conn = make_key("nut_conn", ups_name) + key_ob = make_key("nut_ob", ups_name) + key_rb = make_key("nut_rb", ups_name) + key_batt = make_key("nut_batt", ups_name) + key_load = make_key("nut_load", ups_name) + key_runtime = make_key("nut_rt", ups_name) + + raw = self._run(["upsc", ups_name], timeout=self.config.nut.timeout) + + if not raw or raw.strip() == "": + if self.alerter.should_alert(key_conn, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"USV nicht erreichbar: {ups_name}", + f"Keine Verbindung zu NUT-USV {ups_name}!\n\n" + f"Läuft der Dienst? → systemctl status nut-server\n" + f"upsc {ups_name}", + key_conn, + ) + return + else: + if self.alerter.clear_alert(key_conn): + self.alerter.send_alert( + "INFO", + f"USV wieder erreichbar: {ups_name}", + f"NUT-USV {ups_name} ist wieder erreichbar ✅", + ) + + # Werte parsen + values: dict[str, str] = {} + for line in raw.splitlines(): + if ": " in line: + k, _, v = line.partition(": ") + values[k.strip()] = v.strip() + + status = values.get("ups.status", "") + batt_charge = _to_int(values.get("battery.charge", "100")) + batt_runtime = _to_int(values.get("battery.runtime", "9999")) + ups_load = _to_int(values.get("ups.load", "0")) + + # ── On Battery ─────────────────────────────────────── + if "OB" in status: + runtime_min = batt_runtime // 60 + lb_hint = "\n⚠️ LOW BATTERY – Shutdown droht!" if "LB" in status else "" + if self.alerter.should_alert(key_ob, "CRITICAL"): + self.alerter.send_alert( + "CRITICAL", + f"USV {ups_name}: STROMAUSFALL!", + f"Netzstrom ausgefallen! USV läuft auf Batterie.{lb_hint}\n\n" + f"🔋 Akku: {batt_charge}%\n" + f"⏱️ Restlaufzeit: {runtime_min} min\n" + f"⚡ Last: {ups_load}%\n\n" + f"upsc {ups_name}", + key_ob, + ) + + # Restlaufzeit kritisch + if batt_runtime < self.config.nut.runtime_crit_sec: + if self.alerter.should_alert(key_runtime, "CRITICAL"): + self.alerter.send_alert( + "CRITICAL", + f"USV {ups_name}: Restlaufzeit kritisch!", + f"USV {ups_name}: Nur noch {runtime_min} min Restlaufzeit!\n\n" + f"System sofort herunterfahren:\n" + f"upsmon -c fsd", + key_runtime, + ) + else: + if self.alerter.clear_alert(key_ob): + self.alerter.send_alert( + "INFO", + f"USV {ups_name}: Netzstrom wiederhergestellt ✅", + f"Netzstrom wieder vorhanden. USV {ups_name} " + f"zurück im Normalbetrieb.", + ) + self.alerter.clear_alert(key_runtime) + + # ── Replace Battery ─────────────────────────────────── + if "RB" in status: + if self.alerter.should_alert(key_rb, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"USV {ups_name}: Batterie ersetzen!", + f"USV {ups_name} meldet: Batterie muss ersetzt werden!\n\n" + f"Aktueller Ladestand: {batt_charge}%\n" + f"upsc {ups_name}", + key_rb, + ) + else: + if self.alerter.clear_alert(key_rb): + self.alerter.send_alert( + "INFO", + f"USV {ups_name}: Batterie-Status OK ✅", + f"USV {ups_name}: Replace-Battery Flag wurde aufgehoben.", + ) + + # ── Akkustand (nur im Netzbetrieb) ──────────────────── + if "OB" not in status: + cfg = self.config.nut + if batt_charge <= cfg.batt_crit_pct: + if self.alerter.should_alert(key_batt, "CRITICAL"): + self.alerter.send_alert( + "CRITICAL", + f"USV {ups_name}: Akkustand kritisch!", + f"USV {ups_name}: Akkustand {batt_charge}% " + f"≤ {cfg.batt_crit_pct}%!\n\nAkku lädt nicht korrekt?", + key_batt, + ) + elif batt_charge <= cfg.batt_warn_pct: + if self.alerter.should_alert(key_batt, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"USV {ups_name}: Akkustand niedrig", + f"USV {ups_name}: Akkustand {batt_charge}% " + f"≤ {cfg.batt_warn_pct}%", + key_batt, + ) + else: + if self.alerter.clear_alert(key_batt): + self.alerter.send_alert( + "INFO", + f"USV {ups_name}: Akkustand OK ✅", + f"USV {ups_name}: Akkustand wieder bei {batt_charge}% ✅", + ) + + # ── Last ────────────────────────────────────────────── + if ups_load >= self.config.nut.load_warn_pct: + if self.alerter.should_alert(key_load, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"USV {ups_name}: Hohe Last", + f"USV {ups_name}: Last {ups_load}% " + f"≥ {self.config.nut.load_warn_pct}%", + key_load, + ) + else: + if self.alerter.clear_alert(key_load): + self.alerter.send_alert( + "INFO", + f"USV {ups_name}: Last normal ✅", + f"USV {ups_name}: Last wieder unter " + f"{self.config.nut.load_warn_pct}% ({ups_load}%) ✅", + ) + + self.log.info( + "USV %s: status=%s batt=%d%% runtime=%ds load=%d%%", + ups_name, status, batt_charge, batt_runtime, ups_load, + ) + + +def _to_int(val: str) -> int: + """Parse int from NUT value, stripping decimals and units.""" + try: + return int(float(val.split()[0])) + except (ValueError, IndexError): + return 0 diff --git a/checks/smart.py b/checks/smart.py new file mode 100644 index 0000000..31541f0 --- /dev/null +++ b/checks/smart.py @@ -0,0 +1,256 @@ +from __future__ import annotations + +import json +import os +import re + +from checks.base import Check +from checks.zfs import ZFSCheck, _normalize_device +from state import make_key + + +class SmartCheck(Check): + def run(self) -> None: + disks = self._collect_disks() + self.log.info("--- SMART: %d Disks ---", len(disks)) + for disk in disks: + self._check_disk(disk) + + def _collect_disks(self) -> list[str]: + """Collect all unique disks: pool members + extra_disks from config.""" + seen: set[str] = set() + result: list[str] = [] + + zfs = ZFSCheck(self.config, self.state, self.alerter) + for pool in self.config.zfs.pools: + for disk in zfs.get_pool_disks(pool): + if disk not in seen: + seen.add(disk) + result.append(disk) + + for extra in self.config.smart.extra_disks: + dev = _normalize_device(extra) + if os.path.exists(dev) and dev not in seen: + seen.add(dev) + result.append(dev) + + return result + + def _check_disk(self, disk: str) -> None: + if not os.path.exists(disk): + self.log.warning("SMART: %s nicht vorhanden", disk) + return + if "nvme" in disk: + self._check_nvme(disk) + else: + self._check_sata(disk) + + # ── SATA ────────────────────────────────────────────────── + + def _check_sata(self, disk: str) -> None: + key_fail = make_key("sm_fail", disk) + key_issues = make_key("sm_iss", disk) + key_delta = make_key("sm_dlt", disk) + + raw = self._run( + ["smartctl", "-j", "-A", "-H", disk], + timeout=self.config.smart.timeout, + ) + if not raw: + self.log.warning("SMART: %s nicht lesbar", disk) + return + + try: + d = json.loads(raw) + except json.JSONDecodeError: + self.log.warning("SMART: JSON-Parsing fehlgeschlagen für %s", disk) + return + + passed = d.get("smart_status", {}).get("passed") + attrs = {a["id"]: a for a in d.get("ata_smart_attributes", {}).get("table", [])} + + def attr_raw(aid: int) -> int: + return attrs.get(aid, {}).get("raw", {}).get("value", 0) + + reallocated = attr_raw(5) + pending = attr_raw(197) + uncorr = attr_raw(198) + temp_val = d.get("temperature", {}).get("current") + if temp_val is None: + temp_val = attrs.get(194, attrs.get(190, {})).get("raw", {}).get("value", 0) + temp = int(temp_val) if temp_val else 0 + + # Overall SMART status + if passed is False: + if self.alerter.should_alert(key_fail, "CRITICAL"): + self.alerter.send_alert( + "CRITICAL", + f"SMART FAILED: {disk}", + f"Disk {disk} meldet SMART FAILED!\n\n" + f"Disk ist kurz vor dem Ausfall – sofort ersetzen!\n" + f"smartctl -a {disk}", + key_fail, + ) + return + else: + if self.alerter.clear_alert(key_fail): + self.alerter.send_alert( + "INFO", + f"SMART wieder OK: {disk}", + f"Disk {disk} meldet wieder PASSED ✅", + ) + + # Schwellwert-Checks + cfg = self.config.smart + issues: list[str] = [] + if reallocated > cfg.reallocated_max: + issues.append(f"⚠️ Reallocated Sectors: {reallocated} (max {cfg.reallocated_max})") + if pending > cfg.pending_max: + issues.append(f"⚠️ Pending Sectors: {pending}") + if uncorr > cfg.uncorrectable_max: + issues.append(f"⚠️ Uncorrectable Errors: {uncorr}") + if 0 < temp > cfg.temp_max: + issues.append(f"🌡️ Temperatur: {temp}°C (max {cfg.temp_max}°C)") + + if issues: + if self.alerter.should_alert(key_issues, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"SMART Warnung: {disk}", + f"Disk {disk}:\n\n" + + "\n".join(issues) + + f"\n\nsmartctl -a {disk}", + key_issues, + ) + else: + if self.alerter.clear_alert(key_issues): + self.alerter.send_alert( + "INFO", + f"SMART OK: {disk}", + f"Disk {disk}: alle Werte wieder im grünen Bereich ✅", + ) + + # Delta-Erkennung + prev_r: int = self.state.get(f"{key_delta}_r", -1) + prev_p: int = self.state.get(f"{key_delta}_p", -1) + + if prev_r >= 0 and reallocated > prev_r: + self.alerter.send_alert( + "WARNING", + f"SMART verschlechtert: {disk}", + f"Disk {disk}: Reallocated Sectors gestiegen!\n" + f"{prev_r} → {reallocated}", + ) + if prev_p >= 0 and pending > prev_p: + self.alerter.send_alert( + "CRITICAL", + f"SMART verschlechtert: {disk}", + f"Disk {disk}: Pending Sectors gestiegen!\n" + f"{prev_p} → {pending}\n\n" + f"⚠️ Scrub + Backup sofort prüfen!", + ) + + self.state.set(f"{key_delta}_r", reallocated) + self.state.set(f"{key_delta}_p", pending) + + self.log.info( + "SATA %s: reallocated=%d pending=%d uncorr=%d temp=%d°C", + disk, reallocated, pending, uncorr, temp, + ) + + # ── NVMe ────────────────────────────────────────────────── + + def _check_nvme(self, disk: str) -> None: + key_crit = make_key("nv_crit", disk) + key_spare = make_key("nv_spare", disk) + key_temp = make_key("nv_temp", disk) + key_media = make_key("nv_med", disk) + + raw = self._run( + ["smartctl", "-j", "-a", disk], + timeout=self.config.smart.timeout, + ) + if not raw: + self.log.warning("SMART NVMe: %s nicht lesbar", disk) + return + + try: + d = json.loads(raw) + except json.JSONDecodeError: + self.log.warning("SMART: JSON-Parsing fehlgeschlagen für %s", disk) + return + + log_page = d.get("nvme_smart_health_information_log", {}) + crit_warn = log_page.get("critical_warning", 0) + media_errors = log_page.get("media_errors", 0) + spare = log_page.get("available_spare", 100) + spare_thresh = log_page.get("available_spare_threshold", 10) + pct_used = log_page.get("percentage_used", 0) + temp = int(d.get("temperature", {}).get("current", 0) or 0) + + # Critical Warning Bits + if crit_warn != 0: + if self.alerter.should_alert(key_crit, "CRITICAL"): + self.alerter.send_alert( + "CRITICAL", + f"NVMe Critical Warning: {disk}", + f"NVMe {disk}: Warning Flag 0x{crit_warn:02x}\n\n" + f"Bit 0: Spare under threshold\n" + f"Bit 1: Temperatur over limit\n" + f"Bit 2: Reliability degraded\n" + f"Bit 3: Read-Only gesetzt\n" + f"Bit 4: Backup Device Failed\n\n" + f"smartctl -a {disk}", + key_crit, + ) + else: + if self.alerter.clear_alert(key_crit): + self.alerter.send_alert( + "INFO", + f"NVMe Warning behoben: {disk}", + f"NVMe {disk}: Critical Warning Flag wieder 0 ✅", + ) + + # Available Spare + if spare <= spare_thresh: + if self.alerter.should_alert(key_spare, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"NVMe Spare niedrig: {disk}", + f"NVMe {disk}: Available Spare {spare}% " + f"(Threshold: {spare_thresh}%)\n" + f"Percentage Used: {pct_used}%\n\n" + f"NVMe nähert sich dem Lebensende.", + key_spare, + ) + else: + self.alerter.clear_alert(key_spare) + + # Temperatur + if 0 < temp > self.config.smart.temp_max: + if self.alerter.should_alert(key_temp, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"NVMe Temperatur: {disk}", + f"NVMe {disk}: {temp}°C " + f"(max {self.config.smart.temp_max}°C)", + key_temp, + ) + else: + self.alerter.clear_alert(key_temp) + + # Media Errors (Delta) + prev_media: int = self.state.get(f"{key_media}_prev", -1) + if prev_media >= 0 and media_errors > prev_media: + self.alerter.send_alert( + "WARNING", + f"NVMe Media Errors gestiegen: {disk}", + f"NVMe {disk}: Media Errors " + f"{prev_media} → {media_errors}", + ) + self.state.set(f"{key_media}_prev", media_errors) + + self.log.info( + "NVMe %s: crit=0x%02x media=%d spare=%d%% pct_used=%d%% temp=%d°C", + disk, crit_warn, media_errors, spare, pct_used, temp, + ) diff --git a/checks/zfs.py b/checks/zfs.py new file mode 100644 index 0000000..1475ee3 --- /dev/null +++ b/checks/zfs.py @@ -0,0 +1,182 @@ +from __future__ import annotations + +import re + +from checks.base import Check +from state import make_key + + +class ZFSCheck(Check): + def run(self) -> None: + for pool in self.config.zfs.pools: + self.log.info("--- Pool: %s ---", pool) + self._check_pool(pool) + + def get_pool_disks(self, pool: str) -> list[str]: + """Return normalized block device paths for all disks in a pool.""" + out = self._run( + ["zpool", "status", "-P", pool], + timeout=self.config.zfs.timeout, + ) + if not out: + return [] + + disks: list[str] = [] + for line in out.splitlines(): + m = re.match(r"\s+(/dev/\S+)", line) + if m: + dev = _normalize_device(m.group(1)) + if dev not in disks: + disks.append(dev) + return disks + + # ── Pool checks ─────────────────────────────────────────── + + def _check_pool(self, pool: str) -> None: + key_missing = make_key("pool_miss", pool) + key_health = make_key("pool_hlth", pool) + key_errors = make_key("pool_err", pool) + key_cap = make_key("pool_cap", pool) + + # Erreichbarkeit + out = self._run( + ["zpool", "list", pool], + timeout=self.config.zfs.timeout, + ) + if out is None: + if self.alerter.should_alert(key_missing, "CRITICAL"): + self.alerter.send_alert( + "CRITICAL", + f"Pool verschwunden: {pool}", + f"Pool {pool} ist nicht mehr importiert!\n\n" + f"zpool import\nzpool status", + key_missing, + ) + return + else: + if self.alerter.clear_alert(key_missing): + self.alerter.send_alert( + "INFO", + f"Pool wieder erreichbar: {pool}", + f"Pool {pool} ist wieder da ✅", + ) + + # Health + health = ( + self._run( + ["zpool", "list", "-H", "-o", "health", pool], + timeout=self.config.zfs.timeout, + ) or "" + ).strip() + + if health != "ONLINE": + self.state.set(f"{key_health}_last", health) + if self.alerter.should_alert(key_health, "CRITICAL"): + details = self._pool_status_summary(pool) + self.alerter.send_alert( + "CRITICAL", + f"Pool {pool}: {health}!", + f"Pool {pool} ist {health}!\n\n" + f"
{details}
\n" + f"zpool status -v {pool}", + key_health, + ) + else: + self.state.set(f"{key_health}_last", "ONLINE") + if self.alerter.clear_alert(key_health): + self.alerter.send_alert( + "INFO", + f"Pool {pool}: wieder ONLINE ✅", + f"Pool {pool} ist wieder gesund.", + ) + + # I/O Errors + err_lines = self._parse_io_errors(pool) + if err_lines: + if self.alerter.should_alert(key_errors, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"Pool {pool}: I/O Fehler!", + f"Pool {pool} hat Fehler auf Disks:\n\n" + f"
{err_lines}
\n" + f"→ zpool scrub {pool}", + key_errors, + ) + else: + if self.alerter.clear_alert(key_errors): + self.alerter.send_alert( + "INFO", + f"Pool {pool}: I/O Fehler behoben ✅", + f"Keine I/O Fehler mehr in Pool {pool}.", + ) + + # Kapazität + cap_str = ( + self._run( + ["zpool", "list", "-H", "-o", "cap", pool], + timeout=self.config.zfs.timeout, + ) or "0" + ).strip().rstrip("%") + cap = int(cap_str) if cap_str.isdigit() else 0 + + if cap >= self.config.zfs.crit_pct: + if self.alerter.should_alert(key_cap, "CRITICAL"): + self.alerter.send_alert( + "CRITICAL", + f"Pool {pool}: {cap}% voll!", + f"Pool {pool} ist zu {cap}% voll!\n" + f"Sofort Platz schaffen – ZFS braucht freien Raum!", + key_cap, + ) + elif cap >= self.config.zfs.warn_pct: + if self.alerter.should_alert(key_cap, "WARNING"): + self.alerter.send_alert( + "WARNING", + f"Pool {pool}: {cap}% voll", + f"Pool {pool} ist zu {cap}% voll.", + key_cap, + ) + else: + if self.alerter.clear_alert(key_cap): + self.alerter.send_alert( + "INFO", + f"Pool {pool}: Füllstand OK", + f"Pool {pool} ist wieder unter " + f"{self.config.zfs.warn_pct}% ✅", + ) + + self.log.info("Pool %s: health=%s cap=%d%%", pool, health, cap) + + def _pool_status_summary(self, pool: str) -> str: + out = self._run(["zpool", "status", pool], timeout=self.config.zfs.timeout) or "" + lines = [ + l for l in out.splitlines() + if re.search(r"state:|status:|action:|DEGRADED|FAULTED|REMOVED|UNAVAIL", l) + ] + return "\n".join(lines[:10]) + + def _parse_io_errors(self, pool: str) -> str: + out = self._run( + ["zpool", "status", "-P", pool], + timeout=self.config.zfs.timeout, + ) or "" + lines = [] + for line in out.splitlines(): + m = re.match(r"\s+(/dev/\S+)\s+\S+\s+(\d+)\s+(\d+)\s+(\d+)", line) + if m: + dev, r, w, c = m.group(1), int(m.group(2)), int(m.group(3)), int(m.group(4)) + if r or w or c: + lines.append(f"{dev} R:{r} W:{w} C:{c}") + return "\n".join(lines) + + +# ── Device Normalisierung ───────────────────────────────────── + +def _normalize_device(dev: str) -> str: + import os + resolved = os.path.realpath(dev) + # nvme0n1p1 → nvme0n1 + resolved = re.sub(r"(nvme\d+n\d+)p\d+$", r"\1", resolved) + # sda1 → sda + resolved = re.sub(r"(sd[a-z]+)\d+$", r"\1", resolved) + return resolved diff --git a/config.py b/config.py new file mode 100644 index 0000000..895a7c0 --- /dev/null +++ b/config.py @@ -0,0 +1,124 @@ +from __future__ import annotations + +import socket +import tomllib +from dataclasses import dataclass, field +from pathlib import Path + + +@dataclass +class TelegramConfig: + token: str + chat_id: str + + +@dataclass +class AlertConfig: + critical_repeat_runs: int = 1 + warning_repeat_runs: int = 6 + + +@dataclass +class ZFSConfig: + pools: list[str] + warn_pct: int = 85 + crit_pct: int = 95 + timeout: int = 10 + + +@dataclass +class SmartConfig: + extra_disks: list[str] = field(default_factory=list) + reallocated_max: int = 0 + pending_max: int = 0 + uncorrectable_max: int = 0 + temp_max: int = 55 + timeout: int = 15 + + +@dataclass +class NutConfig: + enabled: bool = False + ups: list[str] = field(default_factory=lambda: ["ups@localhost"]) + batt_warn_pct: int = 50 + batt_crit_pct: int = 20 + load_warn_pct: int = 80 + runtime_crit_sec: int = 300 + timeout: int = 10 + + +@dataclass +class Config: + telegram: TelegramConfig + zfs: ZFSConfig + smart: SmartConfig + nut: NutConfig + alert: AlertConfig + hostname: str = field(default_factory=socket.gethostname) + log_file: str = "/var/log/server-monitor/monitor.log" + state_file: str = "/var/lib/server-monitor/state.json" + + +def load_config(path: str = "/etc/server-monitor.toml") -> Config: + p = Path(path) + if not p.exists(): + raise FileNotFoundError(f"Config fehlt: {path}") + + with open(p, "rb") as f: + raw = tomllib.load(f) + + tg = raw.get("telegram", {}) + if not tg.get("token") or tg["token"] in ("", "DEIN_TOKEN"): + raise ValueError("telegram.token nicht gesetzt") + if not tg.get("chat_id") or tg["chat_id"] in ("", "DEINE_CHAT_ID"): + raise ValueError("telegram.chat_id nicht gesetzt") + telegram = TelegramConfig(token=tg["token"], chat_id=str(tg["chat_id"])) + + zfs_raw = raw.get("zfs", {}) + if not zfs_raw.get("pools"): + raise ValueError("zfs.pools ist leer") + zfs = ZFSConfig( + pools=zfs_raw["pools"], + warn_pct=zfs_raw.get("warn_pct", 85), + crit_pct=zfs_raw.get("crit_pct", 95), + timeout=zfs_raw.get("timeout", 10), + ) + + sm = raw.get("smart", {}) + smart = SmartConfig( + extra_disks=sm.get("extra_disks", []), + reallocated_max=sm.get("reallocated_max", 0), + pending_max=sm.get("pending_max", 0), + uncorrectable_max=sm.get("uncorrectable_max", 0), + temp_max=sm.get("temp_max", 55), + timeout=sm.get("timeout", 15), + ) + + nut_raw = raw.get("nut", {}) + nut = NutConfig( + enabled=nut_raw.get("enabled", False), + ups=nut_raw.get("ups", ["ups@localhost"]), + batt_warn_pct=nut_raw.get("batt_warn_pct", 50), + batt_crit_pct=nut_raw.get("batt_crit_pct", 20), + load_warn_pct=nut_raw.get("load_warn_pct", 80), + runtime_crit_sec=nut_raw.get("runtime_crit_sec", 300), + timeout=nut_raw.get("timeout", 10), + ) + + alert_raw = raw.get("alert", {}) + alert = AlertConfig( + critical_repeat_runs=alert_raw.get("critical_repeat_runs", 1), + warning_repeat_runs=alert_raw.get("warning_repeat_runs", 6), + ) + + mon = raw.get("monitoring", {}) + return Config( + telegram=telegram, + zfs=zfs, + smart=smart, + nut=nut, + alert=alert, + hostname=mon.get("hostname") or socket.gethostname(), + log_file=mon.get("log_file", "/var/log/server-monitor/monitor.log"), + state_file=mon.get("state_file", "/var/lib/server-monitor/state.json"), + ) diff --git a/install.sh b/install.sh new file mode 100644 index 0000000..cf32c0a --- /dev/null +++ b/install.sh @@ -0,0 +1,37 @@ +#!/bin/bash +# server-monitor install.sh +set -euo pipefail + +INSTALL_DIR="/usr/local/lib/server-monitor" +CONFIG_DST="/etc/server-monitor.toml" + +echo "==> Abhängigkeiten installieren..." +apt-get install -y --no-install-recommends smartmontools curl nut-client python3 + +echo "==> Skripte installieren nach ${INSTALL_DIR}..." +mkdir -p "$INSTALL_DIR/checks" +cp monitor.py config.py state.py notify.py "$INSTALL_DIR/" +cp checks/*.py "$INSTALL_DIR/checks/" +chmod +x "${INSTALL_DIR}/monitor.py" + +echo "==> Verzeichnisse anlegen..." +mkdir -p /var/log/server-monitor /var/lib/server-monitor + +echo "==> Config installieren..." +if [[ -f "$CONFIG_DST" ]]; then + echo " Config existiert bereits – wird nicht überschrieben." +else + cp server-monitor.toml "$CONFIG_DST" + echo " ⚠️ Bitte Token + Chat-ID eintragen in: ${CONFIG_DST}" +fi + +echo "==> systemd Units installieren..." +cp systemd/server-monitor.service /etc/systemd/system/ +cp systemd/server-monitor.timer /etc/systemd/system/ +systemctl daemon-reload +systemctl enable --now server-monitor.timer + +echo "" +echo "✅ Fertig! Status prüfen:" +echo " systemctl status server-monitor.timer" +echo " journalctl -u server-monitor.service -f" diff --git a/monitor.py b/monitor.py new file mode 100644 index 0000000..3db73a6 --- /dev/null +++ b/monitor.py @@ -0,0 +1,65 @@ +#!/usr/bin/env python3 +"""server-monitor – Erweitertes Server-Monitoring mit Telegram-Alerts.""" + +from __future__ import annotations + +import logging +import sys +from pathlib import Path + +from config import load_config +from notify import Alerter +from state import State +from checks.zfs import ZFSCheck +from checks.smart import SmartCheck +from checks.nut import NutCheck + +CONFIG_FILE = "/etc/server-monitor.toml" + + +def setup_logging(log_file: str) -> None: + Path(log_file).parent.mkdir(parents=True, exist_ok=True) + logging.basicConfig( + level=logging.INFO, + format="[%(asctime)s] %(message)s", + datefmt="%Y-%m-%d %H:%M:%S", + handlers=[ + logging.FileHandler(log_file), + logging.StreamHandler(sys.stdout), + ], + ) + + +def main() -> None: + try: + config = load_config(CONFIG_FILE) + except (FileNotFoundError, ValueError) as e: + print(f"FEHLER: {e}", file=sys.stderr) + sys.exit(1) + + setup_logging(config.log_file) + log = logging.getLogger("main") + + state = State(config.state_file) + alerter = Alerter(config, state) + + log.info("====== server-monitor Start ======") + + checks = [ + ZFSCheck(config, state, alerter), + SmartCheck(config, state, alerter), + ] + if config.nut.enabled: + checks.append(NutCheck(config, state, alerter)) + + for check in checks: + try: + check.run() + except Exception as e: + log.error("Check %s fehlgeschlagen: %s", check.__class__.__name__, e) + + log.info("====== server-monitor Ende ======") + + +if __name__ == "__main__": + main() diff --git a/notify.py b/notify.py new file mode 100644 index 0000000..d2e767b --- /dev/null +++ b/notify.py @@ -0,0 +1,99 @@ +from __future__ import annotations + +import json +import logging +import time +import urllib.parse +import urllib.request +from datetime import datetime + +from config import Config +from state import State + +log = logging.getLogger(__name__) + +ICONS = {"CRITICAL": "🔴", "WARNING": "🟡", "INFO": "🟢"} + + +class Alerter: + def __init__(self, config: Config, state: State) -> None: + self.config = config + self.state = state + + # ── Persistent Alerting ─────────────────────────────────── + # Gleiche Logik wie im Bash-Script: + # - Neuer Fehler → sofort melden, runs = 0 + # - Fehler bleibt → runs++, melden wenn runs >= repeat_runs + # - Fehler weg → Entwarnung, State löschen + + def should_alert(self, key: str, level: str) -> bool: + repeat = ( + self.config.alert.critical_repeat_runs + if level == "CRITICAL" + else self.config.alert.warning_repeat_runs + ) + active = self.state.get(f"{key}_active") + runs: int = self.state.get(f"{key}_runs", 0) + + if active != "1": + self.state.set(f"{key}_active", "1") + self.state.set(f"{key}_runs", 0) + return True + + runs += 1 + self.state.set(f"{key}_runs", runs) + + if repeat == 0: + return False + if runs >= repeat: + self.state.set(f"{key}_runs", 0) + return True + return False + + def clear_alert(self, key: str) -> bool: + """Clear state and return True if the alert was previously active.""" + was_active = self.state.get(f"{key}_active") == "1" + self.state.delete(f"{key}_active", f"{key}_runs") + return was_active + + # ── Sending ─────────────────────────────────────────────── + + def send_alert(self, level: str, title: str, body: str, key: str = "") -> None: + icon = ICONS.get(level, "⚪") + suffix = "" + if key: + runs: int = self.state.get(f"{key}_runs", 0) + if runs > 0: + suffix = f" (Erinnerung #{runs})" + + ts = datetime.now().strftime("%d.%m.%Y %H:%M:%S") + msg = ( + f"{icon} [{level}] {self.config.hostname}{suffix}\n" + f"{title}\n\n" + f"{body}\n\n" + f"🕐 {ts}" + ) + log.info("ALERT %s: %s", level, title) + self._send_telegram(msg) + + def _send_telegram(self, text: str) -> bool: + url = f"https://api.telegram.org/bot{self.config.telegram.token}/sendMessage" + data = urllib.parse.urlencode({ + "chat_id": self.config.telegram.chat_id, + "text": text, + "parse_mode": "HTML", + }).encode() + + for attempt in range(1, 4): + try: + with urllib.request.urlopen(url, data=data, timeout=10) as resp: + result = json.loads(resp.read()) + if result.get("ok"): + return True + except Exception as e: + log.warning("Telegram Fehler (Versuch %d/3): %s", attempt, e) + if attempt < 3: + time.sleep(attempt * 2) + + log.error("Telegram: Alert konnte nicht gesendet werden!") + return False diff --git a/server-monitor.toml b/server-monitor.toml new file mode 100644 index 0000000..a7c513b --- /dev/null +++ b/server-monitor.toml @@ -0,0 +1,56 @@ +# ============================================================ +# server-monitor Konfiguration +# ============================================================ + +[telegram] +token = "DEIN_TOKEN" +chat_id = "DEINE_CHAT_ID" + +[monitoring] +# hostname = "" # Standard: automatisch via hostname +log_file = "/var/log/server-monitor/monitor.log" +state_file = "/var/lib/server-monitor/state.json" + +# ------------------------------------------------------------ +# Persistent Alerting +# Monitor läuft alle 5 min via systemd timer. +# critical_repeat_runs = 1 → jedes Mal (alle 5 min) +# warning_repeat_runs = 6 → alle 30 min +# 0 = kein Repeat (einmalig) +# ------------------------------------------------------------ +[alert] +critical_repeat_runs = 1 +warning_repeat_runs = 6 + +# ------------------------------------------------------------ +# ZFS +# ------------------------------------------------------------ +[zfs] +pools = ["blackhole", "vault"] +warn_pct = 85 +crit_pct = 95 +timeout = 10 + +# ------------------------------------------------------------ +# SMART +# ------------------------------------------------------------ +[smart] +extra_disks = ["/dev/sdp"] # Disks außerhalb von Pools +reallocated_max = 0 +pending_max = 0 +uncorrectable_max = 0 +temp_max = 55 # °C +timeout = 15 + +# ------------------------------------------------------------ +# NUT / USV +# ------------------------------------------------------------ +[nut] +enabled = false +# Name(n) aus /etc/nut/ups.conf (upsc @) +ups = ["ups@localhost"] +batt_warn_pct = 50 # % Akkuladestand Warnung +batt_crit_pct = 20 # % Akkuladestand Critical +load_warn_pct = 80 # % Last Warnung +runtime_crit_sec = 300 # Sekunden Restlaufzeit bis Critical (5 min) +timeout = 10 diff --git a/state.py b/state.py new file mode 100644 index 0000000..2dbdc99 --- /dev/null +++ b/state.py @@ -0,0 +1,44 @@ +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +from typing import Any + + +def make_key(prefix: str, identifier: str) -> str: + """Stable, collision-safe state key from prefix + arbitrary identifier.""" + h = hashlib.md5(identifier.encode()).hexdigest()[:8] + return f"{prefix}_{h}" + + +class State: + def __init__(self, path: str) -> None: + self._path = Path(path) + self._path.parent.mkdir(parents=True, exist_ok=True) + self._data: dict[str, Any] = {} + self._load() + + def _load(self) -> None: + if self._path.exists(): + try: + self._data = json.loads(self._path.read_text()) + except (json.JSONDecodeError, OSError): + self._data = {} + + def _save(self) -> None: + self._path.write_text(json.dumps(self._data, indent=2)) + + def get(self, key: str, default: Any = None) -> Any: + return self._data.get(key, default) + + def set(self, key: str, value: Any) -> None: + self._data[key] = value + self._save() + + def delete(self, *keys: str) -> None: + changed = any(k in self._data for k in keys) + for k in keys: + self._data.pop(k, None) + if changed: + self._save() diff --git a/systemd/server-monitor.service b/systemd/server-monitor.service new file mode 100644 index 0000000..43fdb38 --- /dev/null +++ b/systemd/server-monitor.service @@ -0,0 +1,10 @@ +[Unit] +Description=Server Monitor +After=network.target + +[Service] +Type=oneshot +ExecStart=/usr/bin/python3 /usr/local/lib/server-monitor/monitor.py +TimeoutStartSec=120 +StandardOutput=journal +StandardError=journal diff --git a/systemd/server-monitor.timer b/systemd/server-monitor.timer new file mode 100644 index 0000000..780feba --- /dev/null +++ b/systemd/server-monitor.timer @@ -0,0 +1,11 @@ +[Unit] +Description=Server Monitor Timer +Requires=server-monitor.service + +[Timer] +OnBootSec=2min +OnUnitActiveSec=5min +AccuracySec=30s + +[Install] +WantedBy=timers.target