diff --git a/monitoring/README.md b/monitoring/README.md new file mode 100644 index 0000000..3cbdd72 --- /dev/null +++ b/monitoring/README.md @@ -0,0 +1,89 @@ +# AC Cluster Monitoring — Deploy Guide + +**Stack:** Prometheus 3.5 · Grafana 12.4 · Alertmanager 0.32 · blackbox_exporter · node_exporter · mysqld_exporter · postgres_exporter +**SPEC:** [ac-docs/specs/20260626_cluster_monitoring.md](../../ac-docs/specs/20260626_cluster_monitoring.md) +**DR:** [ac-docs/DRs/20260626_cluster_monitoring.md](../../ac-docs/DRs/20260626_cluster_monitoring.md) +**Status:** R0 live on cast04 as of 2026-06-26 + +--- + +## Topology + +| Node | IP | Role | Exporters | +|------|----|------|-----------| +| cast04 | 10.7.16.239 | Monitoring host | Prometheus + Grafana + Alertmanager + blackbox + node | +| cast01 | 10.7.16.236 | App + Gitea + PostgreSQL | node + postgres_exporter | +| cast02 | 10.7.16.237 | App | node_exporter | +| cast03 | 10.7.16.238 | App | node_exporter | +| artc0 | 10.7.16.128 | BE + MariaDB + Docker | node + mysqld_exporter (Docker) | + +## Access + +- **Grafana (public):** https://apps.f0xx.org/app/androidcast_project/monitor/ + Login via existing PHP session (auto-SSO). Redirects to `/login` if not authenticated. +- **Grafana (direct internal):** http://10.7.16.239:3000/ (admin/acMonitor2026!) +- **Prometheus UI:** http://10.7.16.239:9090/ (no auth; internal only) +- **Alertmanager UI:** http://10.7.16.239:9093/ (no auth; internal only) + +## Config files (on cast04) + +| File | Purpose | +|------|---------| +| `/etc/prometheus/prometheus.yml` | Scrape targets | +| `/etc/prometheus/rules/node_alerts.yml` | HW/OS alert rules (10 rules) | +| `/etc/prometheus/rules/service_alerts.yml` | Service/HTTP/DB alert rules (10 rules) | +| `/etc/alertmanager/alertmanager.yml` | Alert routing (email + Telegram stub) | +| `/etc/grafana/grafana.ini` | Grafana config (subpath, auth proxy) | +| `/var/lib/grafana/provisioning/` | Auto-provisioned datasource + dashboards dir | +| `/etc/conf.d/grafana` | Grafana runtime overrides (bind address, provisioning path) | + +## Config files (on artc0/BE) + +| File | Purpose | +|------|---------| +| `/etc/nginx/conf.d/apps.conf` | BE nginx — includes monitoring proxy location | +| `/var/www/.../api/grafana-auth-check.php` | PHP session → X-WEBAUTH-USER auth bridge | +| `/etc/mysqld-exporter.cnf` | MariaDB exporter credentials | + +## Services management (cast04) + +```sh +# Status +for svc in prometheus node-exporter blackbox-exporter alertmanager grafana; do + echo "$svc: $(rc-service $svc status 2>&1 | grep -o 'started\|stopped')" +done + +# Restart all +for svc in prometheus node-exporter blackbox-exporter alertmanager grafana; do + sudo rc-service $svc restart +done +``` + +## Grafana dashboards (imported) + +| Dashboard | URL | +|-----------|-----| +| Node Exporter Full | `/d/rYdddlPWk/node-exporter-full` | +| MySQL Exporter Quickstart | `/d/...` | +| Prometheus Blackbox Exporter | `/d/xtkCtBkiz/prometheus-blackbox-exporter` | +| Node Exporter EN Stats | `/d/xfpJB9FGz/...` | + +## TODO before production + +1. **Email alerts:** Fill in Gmail app password in `/etc/alertmanager/alertmanager.yml` +2. **Telegram alerts:** Add `bot_token` + `chat_id` to `multi-critical` receiver +3. **postgres_exporter:** Verify cast01 `pg_up 1` after next restart +4. **UptimeRobot:** Create account + add HTTP monitor for `apps.f0xx.org/app/androidcast_project/` +5. **Grafana admin password:** Change from default `acMonitor2026!` after setting up auth proxy + +## Alert thresholds + +| Metric | Warning | Critical | +|--------|---------|----------| +| CPU | >85% for 10m | >95% for 5m | +| RAM free | <15% for 5m | <5% for 2m | +| Disk free | <20% for 15m | <10% for 10m | +| Load (per vCPU) | >1.5× for 5m | >3.0× for 5m | +| HTTP probe | >5s for 5m | down >3m | +| SSL cert | <14 days | <7 days | +| MariaDB | high connections | down >2m | diff --git a/monitoring/artc0/grafana-auth-check.php b/monitoring/artc0/grafana-auth-check.php new file mode 100644 index 0000000..d923b2e --- /dev/null +++ b/monitoring/artc0/grafana-auth-check.php @@ -0,0 +1,52 @@ + '/app/androidcast_project/', + 'secure' => isset($_SERVER['HTTPS']), + 'httponly' => true, + 'samesite' => 'Lax', + ]); + session_start(); +} + +// Check if user is logged in — Auth.php stores array in $_SESSION['user'] +// with at least { 'id' => int, 'username' => string, ... } +$user = $_SESSION['user'] ?? null; + +if (empty($user) || empty($user['id'])) { + http_response_code(401); + exit; +} + +// Require full login (not just 2FA pending) +if (isset($user['pending_2fa']) && $user['pending_2fa']) { + http_response_code(401); + exit; +} + +$username = (string)($user['username'] ?? $user['email'] ?? 'user_' . $user['id']); + +// Sanitize — Grafana username must be a valid identifier +$grafana_user = preg_replace('/[^a-zA-Z0-9._@-]/', '_', $username); + +http_response_code(200); +header('X-WEBAUTH-USER: ' . $grafana_user); +header('Content-Type: text/plain'); +echo 'ok'; diff --git a/monitoring/artc0/mysqld-exporter-docker.sh b/monitoring/artc0/mysqld-exporter-docker.sh new file mode 100644 index 0000000..26874ba --- /dev/null +++ b/monitoring/artc0/mysqld-exporter-docker.sh @@ -0,0 +1,53 @@ +#!/bin/sh +# Deploy mysqld_exporter as Docker container on artc0 +# Run once as root / sudo after creating the MariaDB monitoring user below. +# +# PREREQUISITE — run in mysql: +# CREATE USER 'exporter'@'localhost' IDENTIFIED BY 'ExporterPass2026!' WITH MAX_USER_CONNECTIONS 3; +# GRANT PROCESS, REPLICATION CLIENT, SELECT ON *.* TO 'exporter'@'localhost'; +# FLUSH PRIVILEGES; +# +# Then run this script. + +set -e + +CONTAINER_NAME="mysqld_exporter" +CNF_FILE="/etc/mysqld-exporter.cnf" + +# Write config file if not present +if [ ! -f "$CNF_FILE" ]; then + cat > "$CNF_FILE" << 'EOF' +[client] +user=exporter +password=ExporterPass2026! +socket=/run/mysqld/mysqld.sock +EOF + chmod 644 "$CNF_FILE" +fi + +# Get socket GID to allow container user to access it +SOCK_GID=$(stat -c '%g' /run/mysqld/mysqld.sock) + +# Stop existing container if running +docker stop "$CONTAINER_NAME" 2>/dev/null || true +docker rm "$CONTAINER_NAME" 2>/dev/null || true + +# Run with Unix socket mount — MariaDB uses skip-networking (no TCP) +docker run -d \ + --name "$CONTAINER_NAME" \ + --restart unless-stopped \ + --user "65534:${SOCK_GID}" \ + -v "${CNF_FILE}:/etc/.my.cnf:ro" \ + -v "/run/mysqld/mysqld.sock:/run/mysqld/mysqld.sock:ro" \ + -p 9104:9104 \ + prom/mysqld-exporter:latest \ + --config.my-cnf=/etc/.my.cnf \ + --collect.info_schema.innodb_metrics \ + --collect.global_status \ + --collect.global_variables \ + --web.listen-address=":9104" + +echo "mysqld_exporter started, listening on :9104" +sleep 3 +docker logs "$CONTAINER_NAME" 2>&1 | tail -5 +curl -s http://localhost:9104/metrics | grep 'mysql_up' diff --git a/monitoring/artc0/nginx-monitor-location.conf b/monitoring/artc0/nginx-monitor-location.conf new file mode 100644 index 0000000..dad41ef --- /dev/null +++ b/monitoring/artc0/nginx-monitor-location.conf @@ -0,0 +1,77 @@ +# Grafana monitoring proxy — add inside the `server { listen 80; }` block in apps.conf +# Place BEFORE the generic location /app/androidcast_project/ block. +# +# Access flow: +# Browser → FE nginx (TLS) → artc0 BE nginx :80 → cast04:3000 (Grafana) +# auth_request validates PHP session on each request. +# ───────────────────────────────────────────────────────────────────────────── + + # ── Internal auth_request endpoint (not directly accessible) ──────────── + location = /app/androidcast_project/api/grafana-auth-check.php { + internal; + include fastcgi_params; + fastcgi_pass unix:/run/php-fpm.socket; + fastcgi_param SCRIPT_FILENAME /var/www/localhost/htdocs/apps/app/androidcast_project/api/grafana-auth-check.php; + fastcgi_param SCRIPT_NAME /app/androidcast_project/api/grafana-auth-check.php; + fastcgi_param REQUEST_URI $request_uri; + # Forward the original session cookie so the PHP script can read it + fastcgi_param HTTP_COOKIE $http_cookie; + } + + # ── Redirect bare paths to trailing slash ──────────────────────────────── + location = /app/androidcast_project/monitor { + return 301 /app/androidcast_project/monitor/; + } + + location = /app/androidcast_project/alertmanager { + return 301 /app/androidcast_project/alertmanager/; + } + + # ── Alertmanager UI at cast04:9093 via auth_request ────────────────────── + location /app/androidcast_project/alertmanager/ { + auth_request /app/androidcast_project/api/grafana-auth-check.php; + auth_request_set $grafana_user $upstream_http_x_webauth_user; + error_page 401 = @monitor_login_redirect; + + proxy_pass http://10.7.16.239:9093/; + proxy_set_header Host $http_host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + + proxy_http_version 1.1; + proxy_read_timeout 120; + proxy_send_timeout 120; + } + + # ── Grafana at cast04:3000 via auth_request ─────────────────────────────── + location /app/androidcast_project/monitor/ { + # Validate PHP session before allowing through + auth_request /app/androidcast_project/api/grafana-auth-check.php; + + # Capture username from auth response header and forward to Grafana + auth_request_set $grafana_user $upstream_http_x_webauth_user; + + # Redirect to login if not authenticated + error_page 401 = @monitor_login_redirect; + + proxy_pass http://10.7.16.239:3000/; + proxy_set_header Host $http_host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + # Pass authenticated user to Grafana auth proxy + proxy_set_header X-WEBAUTH-USER $grafana_user; + + proxy_http_version 1.1; + proxy_read_timeout 300; + proxy_send_timeout 300; + # WebSocket support (Grafana live) + proxy_set_header Upgrade $http_upgrade; + proxy_set_header Connection $connection_upgrade; + } + + # ── Login redirect for unauthenticated monitor access ──────────────────── + location @monitor_login_redirect { + return 302 /app/androidcast_project/login?redirect=$request_uri; + } diff --git a/monitoring/cast04/alert-notifier.env b/monitoring/cast04/alert-notifier.env new file mode 100644 index 0000000..5ae932e --- /dev/null +++ b/monitoring/cast04/alert-notifier.env @@ -0,0 +1,24 @@ +# alert-notifier.env — config for alert-notifier.py on cast04 +# Keep this file readable only by the service user (chmod 600). + +# ── URL shortener ────────────────────────────────────────────────────────── +SHORTENER_INTERNAL=http://10.7.16.128 +SHORTENER_HOST_HDR=s.f0xx.org +SHORTENER_BEARER=3b7031ef38d9a5cb33f6b2e789c27fe8ae32c743f6aa8df7 +SHORTENER_TTL=86400 + +# ── SMTP (Gmail app password) ────────────────────────────────────────────── +SMTP_HOST=smtp.gmail.com +SMTP_PORT=587 +SMTP_USER=bestcastr@gmail.com +SMTP_PASS=wnxv caln tcmt ldpn +EMAIL_FROM=Android Cast Alerts +EMAIL_TO=a.afanasieff@gmail.com + +# ── Telegram ─────────────────────────────────────────────────────────────── +TELEGRAM_BOT=8841511259:AAGNhJICFiyID-3C8LvoMMmnpfw-x6bkVpA +TELEGRAM_CHAT=5616327561 + +# ── Service ──────────────────────────────────────────────────────────────── +LISTEN_PORT=9099 +FALLBACK_URL=https://apps.f0xx.org/app/androidcast_project/alertmanager/ diff --git a/monitoring/cast04/alert-notifier.py b/monitoring/cast04/alert-notifier.py new file mode 100644 index 0000000..438eb41 --- /dev/null +++ b/monitoring/cast04/alert-notifier.py @@ -0,0 +1,371 @@ +#!/usr/bin/env python3 +""" +Alertmanager webhook receiver — alert-notifier.py +Runs on cast04:9099. Receives JSON from Alertmanager, then for each batch: + 1. Shortens the alert URL via the project URL shortener (s.f0xx.org) + 2. Fetches QR PNG from the shortener's built-in QR endpoint + 3. Sends HTML e-mail with the QR embedded as inline image + 4. Sends Telegram message with the QR image attached + +Config is read from alert-notifier.env in the same directory as this script. +""" + +import http.server +import json +import smtplib +import ssl +import subprocess +import urllib.request +import urllib.parse +import email.mime.multipart +import email.mime.text +import email.mime.image +import email.utils +import logging +import os +import sys +import pathlib + +# ── Config defaults (overridden by .env file) ─────────────────────────────── +SHORTENER_INTERNAL = os.environ.get('SHORTENER_INTERNAL', 'http://10.7.16.128') +SHORTENER_HOST_HDR = os.environ.get('SHORTENER_HOST_HDR', 's.f0xx.org') +SHORTENER_BEARER = os.environ.get('SHORTENER_BEARER', '') +SHORTENER_TTL = int(os.environ.get('SHORTENER_TTL', '86400')) # 1 day default + +SMTP_HOST = os.environ.get('SMTP_HOST', 'smtp.gmail.com') +SMTP_PORT = int(os.environ.get('SMTP_PORT', '587')) +SMTP_USER = os.environ.get('SMTP_USER', 'bestcastr@gmail.com') +SMTP_PASS = os.environ.get('SMTP_PASS', '') +EMAIL_FROM = os.environ.get('EMAIL_FROM', 'Android Cast Alerts ') +EMAIL_TO = os.environ.get('EMAIL_TO', 'a.afanasieff@gmail.com') + +TELEGRAM_BOT = os.environ.get('TELEGRAM_BOT', '') +TELEGRAM_CHAT = int(os.environ.get('TELEGRAM_CHAT', '0')) + +LISTEN_PORT = int(os.environ.get('LISTEN_PORT', '9099')) + +# Default URL to shorten when no alert-specific URL is available +FALLBACK_URL = os.environ.get( + 'FALLBACK_URL', + 'https://apps.f0xx.org/app/androidcast_project/alertmanager/' +) + +logging.basicConfig( + level=logging.INFO, + format='%(asctime)s %(levelname)s %(message)s', + stream=sys.stdout, +) +log = logging.getLogger('alert-notifier') + + +# ── URL shortener ──────────────────────────────────────────────────────────── + +def shorten(url: str) -> dict: + """ + POST the destination URL (no ?src=qr) to the project URL shortener. + Returns dict with keys: + short_url – plain https://s.f0xx.org/SLUG (used as click-link in email/TG) + qr_url – https://s.f0xx.org/SLUG?src=qr (encoded into the QR image) + slug – bare slug + Raises on failure. + """ + payload = json.dumps({ + 'url': url, + 'bearer': SHORTENER_BEARER, + 'ttl': SHORTENER_TTL, + }).encode() + req = urllib.request.Request( + f'{SHORTENER_INTERNAL}/api/v1/shorten', + data=payload, + headers={ + 'Content-Type': 'application/json', + 'Host': SHORTENER_HOST_HDR, + }, + method='POST', + ) + with urllib.request.urlopen(req, timeout=8) as resp: + data = json.loads(resp.read()) + if data.get('code', '') != '0': + raise RuntimeError(f"shortener error: {data}") + short_url = data['u'] + slug = short_url.rsplit('/', 1)[-1] + # The QR image encodes the short URL WITH ?src=qr so every mobile scan + # is counted separately from normal link clicks on the backend. + qr_url = short_url + ('&src=qr' if '?' in short_url else '?src=qr') + return { + 'short_url': short_url, + 'qr_url': qr_url, + 'slug': slug, + } + + +def make_qr_png(url: str) -> bytes: + """ + Generate a QR code PNG for *url* using the system qrencode binary. + Returns raw PNG bytes. Raises on failure. + """ + result = subprocess.run( + ['qrencode', '-t', 'PNG', '-o', '-', '-s', '6', '--', url], + capture_output=True, + check=True, + ) + return result.stdout + + +# ── E-mail ─────────────────────────────────────────────────────────────────── + +SEVERITY_EMOJI = { + 'critical': '🔴', + 'warning': '🟡', + 'info': '🔵', +} + +STATUS_EMOJI = { + 'firing': '🔥', + 'resolved': '✅', +} + + +def _alert_html_row(alert: dict) -> str: + status = alert.get('status', 'firing') + labels = alert.get('labels', {}) + anns = alert.get('annotations', {}) + name = labels.get('alertname', '(unknown)') + instance = labels.get('instance', '') + severity = labels.get('severity', '') + summary = anns.get('summary', name) + desc = anns.get('description', '') + starts = alert.get('startsAt', '')[:19].replace('T', ' ') + s_emoji = STATUS_EMOJI.get(status, '') + sv_emoji = SEVERITY_EMOJI.get(severity, '') + color = '#c0392b' if severity == 'critical' else '#e67e22' if severity == 'warning' else '#2980b9' + return f""" + + + {s_emoji} {sv_emoji} {status.upper()} + — {name} + {f' @ {instance}' if instance else ''} + + + + + {f'Summary: {summary}
' if summary else ''} + {f'Detail: {desc}
' if desc else ''} + {f'Fired: {starts} UTC' if starts else ''} + + """ + + +def send_email(subject: str, alerts: list, short_url: str, qr_png: bytes) -> None: + msg = email.mime.multipart.MIMEMultipart('related') + msg['From'] = EMAIL_FROM + msg['To'] = EMAIL_TO + msg['Subject'] = subject + msg['Date'] = email.utils.formatdate(localtime=False) + msg['Message-ID'] = email.utils.make_msgid(domain='cast04.intra.raptor.org') + + rows = ''.join(_alert_html_row(a) for a in alerts) + + html = f""" + +
+

Android Cast — Monitoring Alert

+
+
+ {rows}
+
+

+ View in Alertmanager: + {short_url} +

+

+ Scan QR to open on mobile (tracked via url-shortener): +

+ QR code +
+

+ cast04 monitoring · Grafana +

+""" + + alternative = email.mime.multipart.MIMEMultipart('alternative') + alternative.attach(email.mime.text.MIMEText(html, 'html', 'utf-8')) + msg.attach(alternative) + + qr_img = email.mime.image.MIMEImage(qr_png, 'png') + qr_img.add_header('Content-ID', '') + qr_img.add_header('Content-Disposition', 'inline', filename='alert-qr.png') + msg.attach(qr_img) + + ctx = ssl.create_default_context() + with smtplib.SMTP(SMTP_HOST, SMTP_PORT, timeout=15) as s: + s.ehlo() + s.starttls(context=ctx) + s.ehlo() + s.login(SMTP_USER, SMTP_PASS) + s.sendmail(SMTP_USER, EMAIL_TO, msg.as_bytes()) + log.info('email sent to %s subject=%r', EMAIL_TO, subject) + + +# ── Telegram ───────────────────────────────────────────────────────────────── + +def send_telegram(text: str, qr_png: bytes) -> None: + if not TELEGRAM_BOT or not TELEGRAM_CHAT: + return + + import io, email.generator + + boundary = '----WebKitFormBoundary7MA4YWxkTrZu0gW' + + # Build multipart/form-data body manually + body = ( + f'--{boundary}\r\n' + f'Content-Disposition: form-data; name="chat_id"\r\n\r\n' + f'{TELEGRAM_CHAT}\r\n' + f'--{boundary}\r\n' + f'Content-Disposition: form-data; name="caption"\r\n\r\n' + f'{text}\r\n' + f'--{boundary}\r\n' + f'Content-Disposition: form-data; name="parse_mode"\r\n\r\n' + f'HTML\r\n' + f'--{boundary}\r\n' + f'Content-Disposition: form-data; name="photo"; filename="alert-qr.png"\r\n' + f'Content-Type: image/png\r\n\r\n' + ).encode() + qr_png + f'\r\n--{boundary}--\r\n'.encode() + + req = urllib.request.Request( + f'https://api.telegram.org/bot{TELEGRAM_BOT}/sendPhoto', + data=body, + headers={'Content-Type': f'multipart/form-data; boundary={boundary}'}, + ) + with urllib.request.urlopen(req, timeout=15) as resp: + result = json.loads(resp.read()) + if not result.get('ok'): + raise RuntimeError(f"Telegram error: {result}") + log.info('telegram photo sent to chat_id=%s', TELEGRAM_CHAT) + + +# ── HTTP handler ───────────────────────────────────────────────────────────── + +class WebhookHandler(http.server.BaseHTTPRequestHandler): + def log_message(self, fmt, *args): # suppress default access log + log.debug(fmt, *args) + + def do_POST(self): + if self.path != '/webhook': + self.send_response(404) + self.end_headers() + return + + length = int(self.headers.get('Content-Length', 0)) + body = self.rfile.read(length) + try: + payload = json.loads(body) + except json.JSONDecodeError: + self.send_response(400) + self.end_headers() + return + + self.send_response(200) + self.end_headers() + + # Process asynchronously (we already returned 200) + try: + self._handle(payload) + except Exception as exc: + log.error('webhook handler error: %s', exc, exc_info=True) + + def _handle(self, payload: dict) -> None: + alerts = payload.get('alerts', []) + status = payload.get('status', 'firing') + ext_url = payload.get('externalURL', FALLBACK_URL) + + if not alerts: + return + + # ── shorten & QR ──────────────────────────────────────────────────── + try: + short = shorten(ext_url) + # QR image encodes short_url?src=qr — generated locally via qrencode + qr_png = make_qr_png(short['qr_url']) + log.info('shortened %s → %s qr_url=%s (%d B)', + ext_url, short['short_url'], short['qr_url'], len(qr_png)) + except Exception as exc: + log.error('shorten/qr failed: %s', exc) + return + + # ── subject / Telegram text ────────────────────────────────────────── + firing_names = [a['labels'].get('alertname', '?') for a in alerts if a.get('status') == 'firing'] + resolved_names = [a['labels'].get('alertname', '?') for a in alerts if a.get('status') == 'resolved'] + + if status == 'resolved': + subject = f'[RESOLVED] {", ".join(resolved_names or firing_names)}' + else: + subject = f'[ALERT] {", ".join(firing_names or [a["labels"].get("alertname","?") for a in alerts])}' + + tg_lines = [] + for a in alerts: + s = a.get('status', 'firing') + lbl = a.get('labels', {}) + ann = a.get('annotations', {}) + emoji = STATUS_EMOJI.get(s, '') + ' ' + SEVERITY_EMOJI.get(lbl.get('severity',''), '') + tg_lines.append( + f'{emoji} {s.upper()} {lbl.get("alertname","?")} @ {lbl.get("instance","")}\n' + f'{ann.get("summary","")}\n' + f'{ann.get("description","")}' + ) + + tg_text = '\n\n'.join(tg_lines) + f'\n\n🔗 {short["short_url"]}' + + # ── send ───────────────────────────────────────────────────────────── + try: + send_email(subject, alerts, short['short_url'], qr_png) + except Exception as exc: + log.error('email failed: %s', exc) + + try: + send_telegram(tg_text, qr_png) + except Exception as exc: + log.error('telegram failed: %s', exc) + + +# ── Entry point ─────────────────────────────────────────────────────────────── + +def load_env(path: str) -> None: + """Load KEY=VALUE pairs from an .env file into os.environ.""" + try: + with open(path) as f: + for line in f: + line = line.strip() + if not line or line.startswith('#') or '=' not in line: + continue + k, _, v = line.partition('=') + os.environ.setdefault(k.strip(), v.strip()) + except FileNotFoundError: + pass + + +if __name__ == '__main__': + env_file = pathlib.Path(__file__).with_suffix('.env') + load_env(str(env_file)) + + # Re-read globals that may have been overridden by .env + SHORTENER_INTERNAL = os.environ.get('SHORTENER_INTERNAL', SHORTENER_INTERNAL) + SHORTENER_HOST_HDR = os.environ.get('SHORTENER_HOST_HDR', SHORTENER_HOST_HDR) + SHORTENER_BEARER = os.environ.get('SHORTENER_BEARER', SHORTENER_BEARER) + SHORTENER_TTL = int(os.environ.get('SHORTENER_TTL', str(SHORTENER_TTL))) + SMTP_HOST = os.environ.get('SMTP_HOST', SMTP_HOST) + SMTP_PORT = int(os.environ.get('SMTP_PORT', str(SMTP_PORT))) + SMTP_USER = os.environ.get('SMTP_USER', SMTP_USER) + SMTP_PASS = os.environ.get('SMTP_PASS', SMTP_PASS) + EMAIL_FROM = os.environ.get('EMAIL_FROM', EMAIL_FROM) + EMAIL_TO = os.environ.get('EMAIL_TO', EMAIL_TO) + TELEGRAM_BOT = os.environ.get('TELEGRAM_BOT', TELEGRAM_BOT) + TELEGRAM_CHAT = int(os.environ.get('TELEGRAM_CHAT', str(TELEGRAM_CHAT))) + LISTEN_PORT = int(os.environ.get('LISTEN_PORT', str(LISTEN_PORT))) + FALLBACK_URL = os.environ.get('FALLBACK_URL', FALLBACK_URL) + + server = http.server.HTTPServer(('127.0.0.1', LISTEN_PORT), WebhookHandler) + log.info('alert-notifier listening on 127.0.0.1:%d', LISTEN_PORT) + server.serve_forever() diff --git a/monitoring/cast04/alertmanager.yml b/monitoring/cast04/alertmanager.yml new file mode 100644 index 0000000..d389255 --- /dev/null +++ b/monitoring/cast04/alertmanager.yml @@ -0,0 +1,61 @@ +# Alertmanager config — cast04 (updated 2026-06-26) +# Track A SPEC: ac-docs/specs/20260626_cluster_monitoring.md §7.4 +# +# Outbound notifications are handled by alert-notifier.py (localhost:9099), +# which shortens URLs via s.f0xx.org, generates QR codes, and sends +# HTML email + Telegram with the QR image attached. + +global: + # No direct SMTP here — alert-notifier handles email delivery. + # smtp_* kept as comment for reference: + # smtp_smarthost: 'smtp.gmail.com:587' + # smtp_auth_username: 'bestcastr@gmail.com' + # smtp_auth_password: 'wnxv caln tcmt ldpn' + +templates: + - '/etc/alertmanager/templates/*.tmpl' + +route: + group_by: ['alertname', 'instance', 'severity'] + group_wait: 30s + group_interval: 5m + repeat_interval: 4h + receiver: 'webhook-default' + + routes: + - match: + severity: critical + receiver: 'webhook-critical' + group_wait: 10s + repeat_interval: 1h + + - match: + severity: warning + receiver: 'webhook-default' + repeat_interval: 6h + + - match: + alertname: Watchdog + receiver: 'null' + +receivers: + - name: 'null' + + - name: 'webhook-default' + webhook_configs: + - url: 'http://127.0.0.1:9099/webhook' + send_resolved: true + max_alerts: 10 + + - name: 'webhook-critical' + webhook_configs: + - url: 'http://127.0.0.1:9099/webhook' + send_resolved: true + max_alerts: 10 + +inhibit_rules: + - source_match: + severity: 'critical' + target_match: + severity: 'warning' + equal: ['instance', 'alertname'] diff --git a/monitoring/cast04/grafana-provisioning/dashboards/default.yaml b/monitoring/cast04/grafana-provisioning/dashboards/default.yaml new file mode 100644 index 0000000..46e1ce9 --- /dev/null +++ b/monitoring/cast04/grafana-provisioning/dashboards/default.yaml @@ -0,0 +1,13 @@ +# Dashboard provisioning config — Grafana reads JSON dashboards from this folder +apiVersion: 1 + +providers: + - name: 'ac-cluster' + orgId: 1 + type: file + disableDeletion: false + updateIntervalSeconds: 30 + allowUiUpdates: true + options: + path: /etc/grafana/provisioning/dashboards + foldersFromFilesStructure: false diff --git a/monitoring/cast04/grafana-provisioning/datasources/prometheus.yaml b/monitoring/cast04/grafana-provisioning/datasources/prometheus.yaml new file mode 100644 index 0000000..7dcd49d --- /dev/null +++ b/monitoring/cast04/grafana-provisioning/datasources/prometheus.yaml @@ -0,0 +1,12 @@ +# Auto-provision Prometheus datasource on first Grafana start +apiVersion: 1 + +datasources: + - name: Prometheus + type: prometheus + access: proxy + url: http://localhost:9090 + isDefault: true + editable: false + jsonData: + timeInterval: "15s" diff --git a/monitoring/cast04/grafana.ini b/monitoring/cast04/grafana.ini new file mode 100644 index 0000000..c506dc6 --- /dev/null +++ b/monitoring/cast04/grafana.ini @@ -0,0 +1,89 @@ +# Grafana config — cast04 +# Served at: https://apps.f0xx.org/app/androidcast_project/monitor/ +# Auth: nginx auth_request validates PHP session → X-WEBAUTH-USER header → Grafana auth proxy + +[DEFAULT] + +[paths] +data = /var/lib/grafana +logs = /var/log/grafana +plugins = /var/lib/grafana/plugins +provisioning = /etc/grafana/provisioning + +[server] +protocol = http +http_addr = 0.0.0.0 +http_port = 3000 +domain = apps.f0xx.org +root_url = https://apps.f0xx.org/app/androidcast_project/monitor/ +serve_from_sub_path = true +enable_gzip = true + +[database] +type = sqlite3 +path = grafana.db + +[session] +provider = file + +[analytics] +reporting_enabled = false +check_for_updates = true + +[security] +# Change after first login; used for direct access bypassing the PHP session +admin_user = admin +admin_password = acMonitor2026! +secret_key = acMonitorSecret2026changeme +disable_initial_admin_creation = false +allow_embedding = true + +[users] +allow_sign_up = false +allow_org_create = false +auto_assign_org = true +auto_assign_org_role = Viewer +default_theme = dark + +# ── Auth proxy (PHP session integration) ──────────────────────────────────── +# nginx validates the PHP session cookie and sets X-WEBAUTH-USER if valid. +# Grafana trusts this header and auto-creates/logs in the user. +[auth.proxy] +enabled = true +header_name = X-WEBAUTH-USER +header_property = username +auto_sign_up = true +# Only trust the header from artc0 nginx (10.7.16.128): +whitelist = 10.7.16.128 +# Sync roles from header (optional, set X-WEBAUTH-ROLE in nginx if needed): +# headers = Role:X-WEBAUTH-ROLE + +[auth.anonymous] +enabled = false + +[auth.basic] +# Keep enabled for direct admin access (curl / provisioning scripts) +enabled = true + +[smtp] +enabled = false # Grafana own SMTP not needed; alertmanager handles mail + +[log] +mode = file +level = info + +[log.file] +log_rotate = true +max_lines = 1000000 +max_size_shift = 28 +daily_rotate = true +max_days = 7 + +[alerting] +enabled = false # Using Prometheus Alertmanager, not Grafana's legacy alerting + +[unified_alerting] +enabled = false # Unified alerting also off; Alertmanager is the source of truth + +[feature_toggles] +enable = diff --git a/monitoring/cast04/prometheus.yml b/monitoring/cast04/prometheus.yml new file mode 100644 index 0000000..6a756f4 --- /dev/null +++ b/monitoring/cast04/prometheus.yml @@ -0,0 +1,105 @@ +# Prometheus main config — cast04 (10.7.16.239) +# Track A monitoring SPEC: ac-docs/specs/20260626_cluster_monitoring.md +# Generated: 2026-06-26 + +global: + scrape_interval: 15s + evaluation_interval: 15s + external_labels: + cluster: 'ac-cluster0' + monitor: 'cast04' + +rule_files: + - "/etc/prometheus/rules/*.yml" + +alerting: + alertmanagers: + - static_configs: + - targets: ['localhost:9093'] + +scrape_configs: + + # ── node_exporter on all cluster nodes ────────────────────────────────────── + - job_name: 'node' + static_configs: + - targets: + - '10.7.16.236:9100' # cast01 + - '10.7.16.237:9100' # cast02 + - '10.7.16.238:9100' # cast03 + - '10.7.16.239:9100' # cast04 (self) + - '10.7.16.128:9100' # artc0 (BE/MariaDB) + labels: + env: 'production' + relabel_configs: + - source_labels: [__address__] + regex: '(10\.7\.16\.236):.*' + target_label: instance + replacement: 'cast01' + - source_labels: [__address__] + regex: '(10\.7\.16\.237):.*' + target_label: instance + replacement: 'cast02' + - source_labels: [__address__] + regex: '(10\.7\.16\.238):.*' + target_label: instance + replacement: 'cast03' + - source_labels: [__address__] + regex: '(10\.7\.16\.239):.*' + target_label: instance + replacement: 'cast04' + - source_labels: [__address__] + regex: '(10\.7\.16\.128):.*' + target_label: instance + replacement: 'artc0' + + # ── blackbox_exporter — HTTP probes (internal BE; FE port 443 unreachable from cluster) ─ + # External HTTPS availability → UptimeRobot (SaaS external probe) + - job_name: 'blackbox_http' + metrics_path: /probe + params: + module: [http_2xx] + static_configs: + - targets: + - 'http://10.7.16.128/app/androidcast_project/' # BE main app + - 'http://10.7.16.128/app/androidcast_project/issues/' # crash reporter + - 'http://10.7.16.239:3000/api/health' # Grafana health + labels: + env: 'production' + relabel_configs: + - source_labels: [__address__] + target_label: __param_target + - source_labels: [__param_target] + target_label: instance + - target_label: __address__ + replacement: 'localhost:9115' + + # TCP probes removed: MariaDB uses skip-networking (socket only), PostgreSQL binds loopback only. + # Database health is covered by mysqld_exporter (job: mysqld) and postgres_exporter (job: postgres). + + # ── MariaDB exporter (artc0) ───────────────────────────────────────────────── + - job_name: 'mysqld' + static_configs: + - targets: ['10.7.16.128:9104'] + labels: + instance: 'artc0-mariadb' + + # ── PostgreSQL exporter (cast01) ───────────────────────────────────────────── + - job_name: 'postgres' + static_configs: + - targets: ['10.7.16.236:9187'] + labels: + instance: 'cast01-postgres' + + # ── prometheus self-scrape ─────────────────────────────────────────────────── + - job_name: 'prometheus' + static_configs: + - targets: ['localhost:9090'] + labels: + instance: 'cast04-prometheus' + + # ── alertmanager self-scrape ───────────────────────────────────────────────── + - job_name: 'alertmanager' + static_configs: + - targets: ['localhost:9093'] + labels: + instance: 'cast04-alertmanager' diff --git a/monitoring/cast04/rules/node_alerts.yml b/monitoring/cast04/rules/node_alerts.yml new file mode 100644 index 0000000..f2feac8 --- /dev/null +++ b/monitoring/cast04/rules/node_alerts.yml @@ -0,0 +1,113 @@ +# Node-level alert rules — hardware / OS +# Thresholds per SPEC ac-docs/specs/20260626_cluster_monitoring.md §4.1 + +groups: + - name: node_health + rules: + + # ── instance down ──────────────────────────────────────────────────────── + - alert: InstanceDown + expr: up{job="node"} == 0 + for: 2m + labels: + severity: critical + annotations: + summary: "Instance {{ $labels.instance }} is down" + description: "node_exporter on {{ $labels.instance }} has been unreachable for >2 min." + + # ── CPU ───────────────────────────────────────────────────────────────── + - alert: HighCpuWarning + expr: > + 100 - (avg by(instance)(rate(node_cpu_seconds_total{mode="idle"}[5m])) * 100) > 85 + for: 10m + labels: + severity: warning + annotations: + summary: "High CPU on {{ $labels.instance }}" + description: "CPU usage is {{ $value | printf \"%.1f\" }}% (threshold: 85%) on {{ $labels.instance }}." + + - alert: HighCpuCritical + expr: > + 100 - (avg by(instance)(rate(node_cpu_seconds_total{mode="idle"}[5m])) * 100) > 95 + for: 5m + labels: + severity: critical + annotations: + summary: "Critical CPU on {{ $labels.instance }}" + description: "CPU usage is {{ $value | printf \"%.1f\" }}% (threshold: 95%) on {{ $labels.instance }}." + + # ── memory ────────────────────────────────────────────────────────────── + - alert: LowMemoryWarning + expr: > + (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) * 100 < 15 + for: 5m + labels: + severity: warning + annotations: + summary: "Low memory on {{ $labels.instance }}" + description: "Available RAM is {{ $value | printf \"%.1f\" }}% (threshold: 15%) on {{ $labels.instance }}." + + - alert: LowMemoryCritical + expr: > + (node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) * 100 < 5 + for: 2m + labels: + severity: critical + annotations: + summary: "Critical memory on {{ $labels.instance }}" + description: "Available RAM is {{ $value | printf \"%.1f\" }}% (threshold: 5%) on {{ $labels.instance }}." + + # ── disk ──────────────────────────────────────────────────────────────── + - alert: LowDiskWarning + expr: > + (node_filesystem_avail_bytes{fstype!~"tmpfs|devtmpfs|overlay"} / + node_filesystem_size_bytes{fstype!~"tmpfs|devtmpfs|overlay"}) * 100 < 20 + for: 15m + labels: + severity: warning + annotations: + summary: "Low disk on {{ $labels.instance }} mountpoint {{ $labels.mountpoint }}" + description: "Free disk is {{ $value | printf \"%.1f\" }}% on {{ $labels.instance }}:{{ $labels.mountpoint }} (threshold: 20%)." + + - alert: LowDiskCritical + expr: > + (node_filesystem_avail_bytes{fstype!~"tmpfs|devtmpfs|overlay"} / + node_filesystem_size_bytes{fstype!~"tmpfs|devtmpfs|overlay"}) * 100 < 10 + for: 10m + labels: + severity: critical + annotations: + summary: "Critical disk on {{ $labels.instance }}:{{ $labels.mountpoint }}" + description: "Free disk is {{ $value | printf \"%.1f\" }}% (threshold: 10%) on {{ $labels.instance }}:{{ $labels.mountpoint }}." + + # ── load ──────────────────────────────────────────────────────────────── + - alert: HighLoadWarning + expr: > + node_load1 / on(instance) group_left() count(node_cpu_seconds_total{mode="idle"}) by (instance) > 1.5 + for: 5m + labels: + severity: warning + annotations: + summary: "High load on {{ $labels.instance }}" + description: "Load average 1m is {{ $value | printf \"%.2f\" }}x per vCPU (threshold: 1.5×) on {{ $labels.instance }}." + + - alert: HighLoadCritical + expr: > + node_load1 / on(instance) group_left() count(node_cpu_seconds_total{mode="idle"}) by (instance) > 3.0 + for: 5m + labels: + severity: critical + annotations: + summary: "Critical load on {{ $labels.instance }}" + description: "Load average 1m is {{ $value | printf \"%.2f\" }}x per vCPU (threshold: 3.0×) on {{ $labels.instance }}." + + # ── network anomaly ───────────────────────────────────────────────────── + - alert: NetworkReceiveSaturation + expr: > + rate(node_network_receive_bytes_total{device!~"lo|dummy.*"}[5m]) > 100 * 1024 * 1024 + for: 5m + labels: + severity: warning + annotations: + summary: "High inbound traffic on {{ $labels.instance }}" + description: "{{ $labels.device }} on {{ $labels.instance }} receiving >100 MB/s for 5 min." diff --git a/monitoring/cast04/rules/service_alerts.yml b/monitoring/cast04/rules/service_alerts.yml new file mode 100644 index 0000000..6fcc503 --- /dev/null +++ b/monitoring/cast04/rules/service_alerts.yml @@ -0,0 +1,87 @@ +# Service-level alert rules — HTTP probes, DB +# Alpine uses OpenRC (not systemd) — nginx availability covered by blackbox HTTP probes +# Thresholds per SPEC ac-docs/specs/20260626_cluster_monitoring.md §4.2–4.3 + +groups: + - name: blackbox_probes + rules: + + # ── HTTP endpoint down ─────────────────────────────────────────────────── + - alert: HttpEndpointDown + expr: probe_success{job="blackbox_http"} == 0 + for: 3m + labels: + severity: critical + annotations: + summary: "HTTP endpoint down: {{ $labels.instance }}" + description: "blackbox probe to {{ $labels.instance }} is failing for >3 min." + + - alert: HttpSlowResponse + expr: probe_duration_seconds{job="blackbox_http"} > 5 + for: 5m + labels: + severity: warning + annotations: + summary: "Slow HTTP response: {{ $labels.instance }}" + description: "HTTP probe to {{ $labels.instance }} takes >5s (current: {{ $value | printf \"%.1f\" }}s)." + + - alert: SslCertExpiringSoon + expr: probe_ssl_earliest_cert_expiry{job="blackbox_http"} - time() < 14 * 86400 + for: 1h + labels: + severity: warning + annotations: + summary: "SSL cert expiring soon: {{ $labels.instance }}" + description: "TLS certificate for {{ $labels.instance }} expires in {{ $value | humanizeDuration }}." + + - alert: SslCertExpiring7d + expr: probe_ssl_earliest_cert_expiry{job="blackbox_http"} - time() < 7 * 86400 + for: 1h + labels: + severity: critical + annotations: + summary: "SSL cert expires in <7 days: {{ $labels.instance }}" + description: "TLS certificate for {{ $labels.instance }} expires in {{ $value | humanizeDuration }} — renew immediately." + + # ── MariaDB ───────────────────────────────────────────────────────────────── + - name: mariadb + rules: + - alert: MariaDBDown + expr: mysql_up == 0 + for: 2m + labels: + severity: critical + annotations: + summary: "MariaDB is down on {{ $labels.instance }}" + description: "mysqld_exporter cannot connect to MariaDB on {{ $labels.instance }}." + + - alert: MariaDBHighConnections + expr: mysql_global_status_threads_connected / mysql_global_variables_max_connections > 0.80 + for: 5m + labels: + severity: warning + annotations: + summary: "MariaDB connection pool high on {{ $labels.instance }}" + description: "{{ $value | printf \"%.0f\" }}% of max_connections used on {{ $labels.instance }}." + + - alert: MariaDBSlowQueries + expr: rate(mysql_global_status_slow_queries[5m]) > 1 + for: 5m + labels: + severity: warning + annotations: + summary: "MariaDB slow queries on {{ $labels.instance }}" + description: "More than 1 slow query/s detected on {{ $labels.instance }}." + + # ── disk I/O saturation ────────────────────────────────────────────────────── + - name: disk_io + rules: + - alert: DiskIoSaturation + expr: > + rate(node_disk_io_time_seconds_total[5m]) > 0.9 + for: 10m + labels: + severity: warning + annotations: + summary: "Disk I/O saturation on {{ $labels.instance }}" + description: "Disk {{ $labels.device }} on {{ $labels.instance }} is >90% I/O utilized for >10 min."