feat(mappa-ntfy-monitor): deploy health monitor mappa → ntfy (#1013)

- /root/mappa-ntfy-monitor/monitor.py (python3 stdlib): GET /health (200 + ok:true + service:mappa), алерт после 3 подряд фейлов раз за эпизод + recovery
- cron */5 в /etc/cron.d/mappa-ntfy-monitor, env 600 (NTFY_* из pass), топик mappa-alerts
- тесты: dry-run OK / негатив 3× → DOWN-алерт / recovery; стек 26 не тронут
- ранбук §Мониторинг живости; источник скрипта в scripts/
This commit is contained in:
2026-08-23 19:38:08 +03:00
parent 56c1cb597c
commit 171ee2e9e6
4 changed files with 203 additions and 22 deletions

View File

@@ -0,0 +1,170 @@
#!/usr/bin/env python3
"""mappa-ntfy-monitor — health check of the mappa service (lives on vds-kzntsv).
Task #1013 (mappa-ntfy-monitor, remainder of mappa #985): alert to ntfy when the
mappa service on VDS goes down. Monitor-only — never touches the mappa stack
(Portainer 26); redeploy/restart is a separate task (infra on VDS via admin).
Check (cron every 5 min via /etc/cron.d/mappa-ntfy-monitor):
GET HEALTH_URL (https://mappa.vds.kzntsv.site/health), expect HTTP 200 and
JSON {"ok": true, "service": "mappa"}. Any of: network error / timeout /
non-200 / bad JSON / ok != true / service != mappa counts as a failure.
Alerting policy (no noise on a single blip):
* consecutive-failure counter persisted in STATE_PATH (JSON);
* after FAIL_THRESHOLD consecutive failures (default 3 = ~15 min) -> one ntfy
alert (Priority high) per outage episode (deduped via the `alerted` flag);
* on recovery -> one ntfy info message, counter reset.
Config: /root/.mappa-ntfy-monitor.env (chmod 600; built from `pass vds-kzntsv/full-env`).
Keys: HEALTH_URL NTFY_URL NTFY_USER NTFY_PASS NTFY_TOPIC FAIL_THRESHOLD STATE_PATH LOG_PATH
"""
import base64
import datetime
import json
import os
import sys
import urllib.error
import urllib.request
DEFAULT_ENV_PATH = "/root/.mappa-ntfy-monitor.env"
DEFAULT_LOG_PATH = "/var/log/mappa-ntfy-monitor.log"
DEFAULT_STATE_PATH = "/root/.mappa-ntfy-monitor.state"
DEFAULT_HEALTH_URL = "https://mappa.vds.kzntsv.site/health"
DEFAULT_NTFY_URL = "https://ntfy.vds.kzntsv.site"
DEFAULT_NTFY_TOPIC = "mappa-alerts"
DEFAULT_FAIL_THRESHOLD = 3
HEALTH_TIMEOUT = 15
NTFY_TIMEOUT = 15
def load_env(path):
cfg = {}
with open(path, encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line or line.startswith("#"):
continue
k, _, v = line.partition("=")
cfg[k.strip()] = v.strip()
return cfg
def log(cfg, msg):
ts = datetime.datetime.now().astimezone().strftime("%Y-%m-%dT%H:%M:%S%z")
line = f"{ts} {msg}"
print(line, flush=True)
try:
with open(cfg["LOG_PATH"], "a", encoding="utf-8") as f:
f.write(line + "\n")
except Exception as e:
print(f"log write failed: {e}", flush=True)
def load_state(path):
try:
with open(path, encoding="utf-8") as f:
data = json.load(f)
return {"fail_count": int(data.get("fail_count", 0)),
"alerted": bool(data.get("alerted", False))}
except Exception:
return {"fail_count": 0, "alerted": False}
def save_state(path, state):
tmp = path + ".tmp"
with open(tmp, "w", encoding="utf-8") as f:
json.dump(state, f)
os.replace(tmp, path)
def health_ok(cfg):
"""Return (ok: bool, detail: str)."""
req = urllib.request.Request(cfg["HEALTH_URL"], method="GET")
try:
with urllib.request.urlopen(req, timeout=HEALTH_TIMEOUT) as r:
status = r.status
raw = r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
return False, f"HTTP {e.code}"
except Exception as e:
return False, f"network: {e}"
if status != 200:
return False, f"HTTP {status}"
try:
data = json.loads(raw)
except Exception as e:
return False, f"bad json: {raw[:80]!r}"
if data.get("ok") is not True:
return False, f"ok != true: {raw[:80]!r}"
if data.get("service") != "mappa":
return False, f"service != mappa: {raw[:80]!r}"
return True, raw
def ntfy_publish(cfg, title, body, priority):
try:
url = f"{cfg['NTFY_URL']}/{cfg['NTFY_TOPIC']}"
req = urllib.request.Request(url, data=body.encode(), method="POST")
token = base64.b64encode(f"{cfg['NTFY_USER']}:{cfg['NTFY_PASS']}".encode()).decode()
req.add_header("Authorization", "Basic " + token)
# urllib encodes header VALUES as latin-1 -> keep title ASCII (emoji goes in body)
req.add_header("Title", title.encode("ascii", "replace").decode("ascii"))
req.add_header("Priority", priority)
with urllib.request.urlopen(req, timeout=NTFY_TIMEOUT) as resp:
log(cfg, f"ntfy publish OK (priority={priority}): HTTP {resp.status}")
return True
except Exception as e:
log(cfg, f"ntfy publish FAILED: {e}")
return False
def main():
env_path = os.environ.get("MAPPA_MONITOR_ENV", DEFAULT_ENV_PATH)
cfg = load_env(env_path)
cfg.setdefault("HEALTH_URL", DEFAULT_HEALTH_URL)
cfg.setdefault("NTFY_URL", DEFAULT_NTFY_URL)
cfg.setdefault("NTFY_TOPIC", DEFAULT_NTFY_TOPIC)
cfg.setdefault("FAIL_THRESHOLD", str(DEFAULT_FAIL_THRESHOLD))
cfg.setdefault("LOG_PATH", DEFAULT_LOG_PATH)
cfg.setdefault("STATE_PATH", DEFAULT_STATE_PATH)
try:
threshold = int(cfg["FAIL_THRESHOLD"])
except ValueError:
threshold = DEFAULT_FAIL_THRESHOLD
ok, detail = health_ok(cfg)
state = load_state(cfg["STATE_PATH"])
if ok:
if state["alerted"]:
prior = state["fail_count"]
state = {"fail_count": 0, "alerted": False}
save_state(cfg["STATE_PATH"], state)
ntfy_publish(cfg, "mappa monitor: RECOVERED",
f"mappa health OK again: {detail}", "default")
log(cfg, f"RECOVERED (was down after {prior} consecutive failures)")
else:
state["fail_count"] = 0
save_state(cfg["STATE_PATH"], state)
log(cfg, f"OK health: {detail}")
return 0
state["fail_count"] += 1
fails = state["fail_count"]
if fails >= threshold and not state["alerted"]:
state["alerted"] = True
save_state(cfg["STATE_PATH"], state)
body = (f"mappa DOWN ({fails} consecutive failures)\n"
f"url={cfg['HEALTH_URL']}\nreason={detail}\n"
f"ts={datetime.datetime.now().astimezone().isoformat()}")
ntfy_publish(cfg, "mappa monitor: DOWN", body, "high")
log(cfg, f"ALERT: mappa down after {fails} consecutive failures ({detail})")
return 1
save_state(cfg["STATE_PATH"], state)
log(cfg, f"FAIL health ({fails}/{threshold}): {detail}")
return 1
if __name__ == "__main__":
sys.exit(main())