Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c64853f643 | ||
|
|
42cf754796 | ||
|
|
28fd5aad14 | ||
|
|
65536d58c5 | ||
|
|
0e852ebd11 | ||
|
|
8aaaa89f03 | ||
|
|
7467236ffd |
@@ -0,0 +1,21 @@
|
|||||||
|
{
|
||||||
|
"env": {
|
||||||
|
"browser": true,
|
||||||
|
"es2021": true
|
||||||
|
},
|
||||||
|
"globals": {
|
||||||
|
"lt": "readonly",
|
||||||
|
"GANDALF_CONFIG": "readonly",
|
||||||
|
"CSS": "readonly"
|
||||||
|
},
|
||||||
|
"rules": {
|
||||||
|
"no-undef": "error",
|
||||||
|
"no-unused-vars": ["warn", { "argsIgnorePattern": "^_", "varsIgnorePattern": "^_" }],
|
||||||
|
"no-console": "off",
|
||||||
|
"eqeqeq": ["error", "always", { "null": "ignore" }]
|
||||||
|
},
|
||||||
|
"parserOptions": {
|
||||||
|
"ecmaVersion": 2021,
|
||||||
|
"sourceType": "script"
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -17,7 +17,7 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
apt-get update -qq
|
apt-get update -qq
|
||||||
apt-get install -y -qq python3 python3-pip
|
apt-get install -y -qq python3 python3-pip
|
||||||
pip3 install flake8
|
pip3 install --break-system-packages flake8
|
||||||
|
|
||||||
- name: Run flake8
|
- name: Run flake8
|
||||||
run: flake8 . --exclude=__pycache__,.git
|
run: flake8 . --exclude=__pycache__,.git
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
apt-get update -qq
|
apt-get update -qq
|
||||||
apt-get install -y -qq python3 python3-pip
|
apt-get install -y -qq python3 python3-pip
|
||||||
pip3 install bandit
|
pip3 install --break-system-packages bandit
|
||||||
|
|
||||||
- name: Run bandit
|
- name: Run bandit
|
||||||
run: bandit -r . --exclude .git,__pycache__,node_modules -ll
|
run: bandit -r . --exclude .git,__pycache__,node_modules -ll
|
||||||
|
|||||||
@@ -17,8 +17,8 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
apt-get update -qq
|
apt-get update -qq
|
||||||
apt-get install -y -qq python3 python3-pip
|
apt-get install -y -qq python3 python3-pip
|
||||||
pip3 install pytest pytest-cov
|
pip3 install --break-system-packages pytest pytest-cov
|
||||||
pip3 install -r requirements.txt --quiet
|
pip3 install --break-system-packages -r requirements.txt --quiet
|
||||||
|
|
||||||
- name: Run pytest with coverage
|
- name: Run pytest with coverage
|
||||||
run: python3 -m pytest tests/ -v --cov=. --cov-report=term-missing --cov-config=.coveragerc
|
run: python3 -m pytest tests/ -v --cov=. --cov-report=term-missing --cov-config=.coveragerc
|
||||||
|
|||||||
@@ -148,7 +148,7 @@ returned by the UniFi API for that device.
|
|||||||
|
|
||||||
| Condition | Priority |
|
| Condition | Priority |
|
||||||
|---|---|
|
|---|---|
|
||||||
| UniFi device offline (≥2 consecutive checks) | P2 High |
|
| UniFi device offline (≥`unifi_failure_threshold` consecutive checks) | P2 High |
|
||||||
| Proxmox host NIC link-down regression (≥2 consecutive checks) | P2 High |
|
| Proxmox host NIC link-down regression (≥2 consecutive checks) | P2 High |
|
||||||
| Host unreachable via ping (≥2 consecutive checks) | P2 High |
|
| Host unreachable via ping (≥2 consecutive checks) | P2 High |
|
||||||
| ≥3 hosts simultaneously reporting interface failures | P1 Critical |
|
| ≥3 hosts simultaneously reporting interface failures | P1 Critical |
|
||||||
@@ -218,6 +218,7 @@ Shared by both processes. Located in the working directory (`/var/www/html/prod/
|
|||||||
"monitor": {
|
"monitor": {
|
||||||
"poll_interval": 120,
|
"poll_interval": 120,
|
||||||
"failure_threshold": 2,
|
"failure_threshold": 2,
|
||||||
|
"unifi_failure_threshold": 5,
|
||||||
"cluster_threshold": 3,
|
"cluster_threshold": 3,
|
||||||
"ping_hosts": [
|
"ping_hosts": [
|
||||||
{ "name": "pbs", "ip": "10.10.10.3" }
|
{ "name": "pbs", "ip": "10.10.10.3" }
|
||||||
@@ -243,6 +244,7 @@ Shared by both processes. Located in the working directory (`/var/www/html/prod/
|
|||||||
| `hosts` | Maps Prometheus instance labels → display hostnames |
|
| `hosts` | Maps Prometheus instance labels → display hostnames |
|
||||||
| `monitor.poll_interval` | Seconds between full check cycles (default: 120) |
|
| `monitor.poll_interval` | Seconds between full check cycles (default: 120) |
|
||||||
| `monitor.failure_threshold` | Consecutive failures before creating ticket (default: 2) |
|
| `monitor.failure_threshold` | Consecutive failures before creating ticket (default: 2) |
|
||||||
|
| `monitor.unifi_failure_threshold` | Consecutive failures before ticketing a UniFi device as offline (default: same as `failure_threshold`). Set higher than `failure_threshold` — UniFi switches/APs can take several minutes to rejoin after a firmware-update reboot, so a low threshold cuts tickets for momentary maintenance reboots instead of real failures. |
|
||||||
| `monitor.cluster_threshold` | Hosts with failures to trigger cluster-wide P1 (default: 3) |
|
| `monitor.cluster_threshold` | Hosts with failures to trigger cluster-wide P1 (default: 3) |
|
||||||
| `monitor.ping_hosts` | Hosts checked only by ping (no node_exporter) |
|
| `monitor.ping_hosts` | Hosts checked only by ping (no node_exporter) |
|
||||||
|
|
||||||
|
|||||||
@@ -64,7 +64,7 @@ _diag_rate: dict = {}
|
|||||||
|
|
||||||
|
|
||||||
def _purge_old_jobs_loop():
|
def _purge_old_jobs_loop():
|
||||||
"""Background thread: remove stale diag jobs and run daily event purge."""
|
"""Background thread: remove stale diagnostic jobs and mark stuck ones done."""
|
||||||
while True:
|
while True:
|
||||||
time.sleep(120)
|
time.sleep(120)
|
||||||
cutoff = time.time() - 600
|
cutoff = time.time() - 600
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ import json
|
|||||||
import logging
|
import logging
|
||||||
import threading
|
import threading
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from datetime import datetime, timedelta
|
from datetime import datetime, timedelta, timezone
|
||||||
from typing import Optional
|
from typing import Optional
|
||||||
|
|
||||||
import pymysql
|
import pymysql
|
||||||
@@ -114,12 +114,12 @@ def upsert_event(
|
|||||||
target_detail: str,
|
target_detail: str,
|
||||||
description: str,
|
description: str,
|
||||||
) -> tuple:
|
) -> tuple:
|
||||||
"""Insert or update a network event. Returns (id, is_new, consecutive_failures)."""
|
"""Insert or update a network event. Returns (id, is_new, consecutive_failures, ticket_id)."""
|
||||||
detail = target_detail or ''
|
detail = target_detail or ''
|
||||||
with get_conn() as conn:
|
with get_conn() as conn:
|
||||||
with conn.cursor() as cur:
|
with conn.cursor() as cur:
|
||||||
cur.execute(
|
cur.execute(
|
||||||
"""SELECT id, consecutive_failures FROM network_events
|
"""SELECT id, consecutive_failures, ticket_id FROM network_events
|
||||||
WHERE event_type=%s AND target_name=%s AND target_detail=%s
|
WHERE event_type=%s AND target_name=%s AND target_detail=%s
|
||||||
AND resolved_at IS NULL LIMIT 1""",
|
AND resolved_at IS NULL LIMIT 1""",
|
||||||
(event_type, target_name, detail),
|
(event_type, target_name, detail),
|
||||||
@@ -134,7 +134,7 @@ def upsert_event(
|
|||||||
WHERE id=%s""",
|
WHERE id=%s""",
|
||||||
(new_count, description, existing['id']),
|
(new_count, description, existing['id']),
|
||||||
)
|
)
|
||||||
return existing['id'], False, new_count
|
return existing['id'], False, new_count, existing.get('ticket_id')
|
||||||
else:
|
else:
|
||||||
cur.execute(
|
cur.execute(
|
||||||
"""INSERT INTO network_events
|
"""INSERT INTO network_events
|
||||||
@@ -142,7 +142,7 @@ def upsert_event(
|
|||||||
VALUES (%s, %s, %s, %s, %s, %s)""",
|
VALUES (%s, %s, %s, %s, %s, %s)""",
|
||||||
(event_type, severity, source_type, target_name, detail, description),
|
(event_type, severity, source_type, target_name, detail, description),
|
||||||
)
|
)
|
||||||
return cur.lastrowid, True, 1
|
return cur.lastrowid, True, 1, None
|
||||||
|
|
||||||
|
|
||||||
def resolve_event(event_type: str, target_name: str, target_detail: str = '') -> None:
|
def resolve_event(event_type: str, target_name: str, target_detail: str = '') -> None:
|
||||||
@@ -281,7 +281,7 @@ def create_suppression(
|
|||||||
) -> int:
|
) -> int:
|
||||||
expires_at = None
|
expires_at = None
|
||||||
if expires_minutes:
|
if expires_minutes:
|
||||||
expires_at = datetime.utcnow() + timedelta(minutes=int(expires_minutes))
|
expires_at = datetime.now(timezone.utc) + timedelta(minutes=int(expires_minutes))
|
||||||
with get_conn() as conn:
|
with get_conn() as conn:
|
||||||
with conn.cursor() as cur:
|
with conn.cursor() as cur:
|
||||||
cur.execute(
|
cur.execute(
|
||||||
|
|||||||
+31
-19
@@ -12,7 +12,7 @@ import logging
|
|||||||
import re
|
import re
|
||||||
import shlex
|
import shlex
|
||||||
import time
|
import time
|
||||||
from datetime import datetime
|
from datetime import datetime, timezone
|
||||||
from typing import Dict, List, Optional
|
from typing import Dict, List, Optional
|
||||||
|
|
||||||
import requests
|
import requests
|
||||||
@@ -618,7 +618,7 @@ class LinkStatsCollector:
|
|||||||
return {
|
return {
|
||||||
'hosts': result_hosts,
|
'hosts': result_hosts,
|
||||||
'unifi_switches': unifi_switches,
|
'unifi_switches': unifi_switches,
|
||||||
'updated': datetime.utcnow().strftime('%Y-%m-%d %H:%M:%S UTC'),
|
'updated': datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M:%S UTC'),
|
||||||
}
|
}
|
||||||
|
|
||||||
def _compute_unifi_rates(self, raw: Dict[str, dict], now: float) -> Dict[str, dict]:
|
def _compute_unifi_rates(self, raw: Dict[str, dict], now: float) -> Dict[str, dict]:
|
||||||
@@ -653,7 +653,7 @@ class LinkStatsCollector:
|
|||||||
# Helpers
|
# Helpers
|
||||||
# --------------------------------------------------------------------------
|
# --------------------------------------------------------------------------
|
||||||
def _now_utc() -> str:
|
def _now_utc() -> str:
|
||||||
return datetime.utcnow().strftime('%Y-%m-%d %H:%M:%S UTC')
|
return datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M:%S UTC')
|
||||||
|
|
||||||
|
|
||||||
# --------------------------------------------------------------------------
|
# --------------------------------------------------------------------------
|
||||||
@@ -677,6 +677,11 @@ class NetworkMonitor:
|
|||||||
mon = self.cfg.get('monitor', {})
|
mon = self.cfg.get('monitor', {})
|
||||||
self.poll_interval = mon.get('poll_interval', 120)
|
self.poll_interval = mon.get('poll_interval', 120)
|
||||||
self.fail_thresh = mon.get('failure_threshold', 2)
|
self.fail_thresh = mon.get('failure_threshold', 2)
|
||||||
|
# UniFi gear (switches/APs) can take several minutes to rejoin after a
|
||||||
|
# firmware-update reboot, much longer than a Proxmox host link flap.
|
||||||
|
# Use a separate, more forgiving threshold so those momentary reboots
|
||||||
|
# don't cut tickets, while still catching genuine device failures.
|
||||||
|
self.unifi_fail_thresh = mon.get('unifi_failure_threshold', self.fail_thresh)
|
||||||
self.cluster_thresh = mon.get('cluster_threshold', 3)
|
self.cluster_thresh = mon.get('cluster_threshold', 3)
|
||||||
self.cluster_name = mon.get('cluster_name', CLUSTER_NAME)
|
self.cluster_name = mon.get('cluster_name', CLUSTER_NAME)
|
||||||
|
|
||||||
@@ -728,12 +733,12 @@ class NetworkMonitor:
|
|||||||
db.check_suppressed(suppressions, 'interface', host, iface) or
|
db.check_suppressed(suppressions, 'interface', host, iface) or
|
||||||
db.check_suppressed(suppressions, 'host', host)
|
db.check_suppressed(suppressions, 'host', host)
|
||||||
)
|
)
|
||||||
event_id, is_new, consec = db.upsert_event(
|
event_id, is_new, consec, ticket_id = db.upsert_event(
|
||||||
'interface_down', 'critical', 'prometheus',
|
'interface_down', 'critical', 'prometheus',
|
||||||
host, iface,
|
host, iface,
|
||||||
f'Interface {iface} on {host} went link-down ({_now_utc()})',
|
f'Interface {iface} on {host} went link-down ({_now_utc()})',
|
||||||
)
|
)
|
||||||
if not sup and consec >= self.fail_thresh:
|
if not sup and consec >= self.fail_thresh and not ticket_id:
|
||||||
self._ticket_interface(event_id, host, iface, consec)
|
self._ticket_interface(event_id, host, iface, consec)
|
||||||
|
|
||||||
if host_has_regression:
|
if host_has_regression:
|
||||||
@@ -744,13 +749,13 @@ class NetworkMonitor:
|
|||||||
# Cluster-wide check – only genuine regressions count
|
# Cluster-wide check – only genuine regressions count
|
||||||
if len(hosts_with_regression) >= self.cluster_thresh:
|
if len(hosts_with_regression) >= self.cluster_thresh:
|
||||||
sup = db.check_suppressed(suppressions, 'all', '')
|
sup = db.check_suppressed(suppressions, 'all', '')
|
||||||
event_id, is_new, consec = db.upsert_event(
|
event_id, is_new, consec, ticket_id = db.upsert_event(
|
||||||
'cluster_network_issue', 'critical', 'prometheus',
|
'cluster_network_issue', 'critical', 'prometheus',
|
||||||
self.cluster_name, '',
|
self.cluster_name, '',
|
||||||
f'{len(hosts_with_regression)} hosts reporting simultaneous interface failures: '
|
f'{len(hosts_with_regression)} hosts reporting simultaneous interface failures: '
|
||||||
f'{", ".join(hosts_with_regression)}',
|
f'{", ".join(hosts_with_regression)}',
|
||||||
)
|
)
|
||||||
if not sup and is_new:
|
if not sup and (is_new or not ticket_id):
|
||||||
title = (
|
title = (
|
||||||
f'[{self.cluster_name}][auto][production][issue][network][cluster-wide] '
|
f'[{self.cluster_name}][auto][production][issue][network][cluster-wide] '
|
||||||
f'Multiple hosts reporting interface failures'
|
f'Multiple hosts reporting interface failures'
|
||||||
@@ -804,12 +809,12 @@ class NetworkMonitor:
|
|||||||
name = d['name']
|
name = d['name']
|
||||||
if not d['connected']:
|
if not d['connected']:
|
||||||
sup = db.check_suppressed(suppressions, 'unifi_device', name)
|
sup = db.check_suppressed(suppressions, 'unifi_device', name)
|
||||||
event_id, is_new, consec = db.upsert_event(
|
event_id, is_new, consec, ticket_id = db.upsert_event(
|
||||||
'unifi_device_offline', 'critical', 'unifi',
|
'unifi_device_offline', 'critical', 'unifi',
|
||||||
name, d.get('type', ''),
|
name, d.get('type', ''),
|
||||||
f'UniFi {name} ({d.get("ip","")}) offline ({_now_utc()})',
|
f'UniFi {name} ({d.get("ip","")}) offline ({_now_utc()})',
|
||||||
)
|
)
|
||||||
if not sup and consec >= self.fail_thresh:
|
if not sup and consec >= self.unifi_fail_thresh and not ticket_id:
|
||||||
self._ticket_unifi(event_id, d)
|
self._ticket_unifi(event_id, d)
|
||||||
else:
|
else:
|
||||||
db.resolve_event('unifi_device_offline', name, d.get('type', ''))
|
db.resolve_event('unifi_device_offline', name, d.get('type', ''))
|
||||||
@@ -837,19 +842,19 @@ class NetworkMonitor:
|
|||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
# Ping-only hosts (no node_exporter)
|
# Ping-only hosts (no node_exporter)
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
def _process_ping_hosts(self, suppressions: list) -> None:
|
def _process_ping_hosts(self, suppressions: list, ping_states: Dict[str, bool]) -> None:
|
||||||
for h in self.cfg.get('monitor', {}).get('ping_hosts', []):
|
for h in self.cfg.get('monitor', {}).get('ping_hosts', []):
|
||||||
name, ip = h['name'], h['ip']
|
name, ip = h['name'], h['ip']
|
||||||
reachable = self.pulse.ping(ip)
|
reachable = ping_states.get(name, False)
|
||||||
|
|
||||||
if not reachable:
|
if not reachable:
|
||||||
sup = db.check_suppressed(suppressions, 'host', name)
|
sup = db.check_suppressed(suppressions, 'host', name)
|
||||||
event_id, is_new, consec = db.upsert_event(
|
event_id, is_new, consec, ticket_id = db.upsert_event(
|
||||||
'host_unreachable', 'critical', 'ping',
|
'host_unreachable', 'critical', 'ping',
|
||||||
name, ip,
|
name, ip,
|
||||||
f'Host {name} ({ip}) unreachable via ping ({_now_utc()})',
|
f'Host {name} ({ip}) unreachable via ping ({_now_utc()})',
|
||||||
)
|
)
|
||||||
if not sup and consec >= self.fail_thresh:
|
if not sup and consec >= self.fail_thresh and not ticket_id:
|
||||||
self._ticket_unreachable(event_id, name, ip, consec)
|
self._ticket_unreachable(event_id, name, ip, consec)
|
||||||
else:
|
else:
|
||||||
db.resolve_event('host_unreachable', name, ip)
|
db.resolve_event('host_unreachable', name, ip)
|
||||||
@@ -882,6 +887,7 @@ class NetworkMonitor:
|
|||||||
def _collect_snapshot(
|
def _collect_snapshot(
|
||||||
self, iface_states: Dict[str, Dict[str, bool]],
|
self, iface_states: Dict[str, Dict[str, bool]],
|
||||||
unifi_devices: Optional[List[dict]] = None,
|
unifi_devices: Optional[List[dict]] = None,
|
||||||
|
ping_states: Optional[Dict[str, bool]] = None,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
# Accept pre-fetched devices; fall back to empty list if unavailable
|
# Accept pre-fetched devices; fall back to empty list if unavailable
|
||||||
display_unifi = unifi_devices if unifi_devices is not None else []
|
display_unifi = unifi_devices if unifi_devices is not None else []
|
||||||
@@ -910,7 +916,7 @@ class NetworkMonitor:
|
|||||||
|
|
||||||
for h in self.cfg.get('monitor', {}).get('ping_hosts', []):
|
for h in self.cfg.get('monitor', {}).get('ping_hosts', []):
|
||||||
name, ip = h['name'], h['ip']
|
name, ip = h['name'], h['ip']
|
||||||
reachable = self.pulse.ping(ip, count=1, timeout=2)
|
reachable = (ping_states or {}).get(name, False)
|
||||||
hosts[name] = {
|
hosts[name] = {
|
||||||
'ip': ip,
|
'ip': ip,
|
||||||
'interfaces': {},
|
'interfaces': {},
|
||||||
@@ -921,7 +927,7 @@ class NetworkMonitor:
|
|||||||
return {
|
return {
|
||||||
'hosts': hosts,
|
'hosts': hosts,
|
||||||
'unifi': display_unifi,
|
'unifi': display_unifi,
|
||||||
'updated': datetime.utcnow().isoformat() + 'Z',
|
'updated': datetime.now(timezone.utc).isoformat().replace('+00:00', 'Z'),
|
||||||
}
|
}
|
||||||
|
|
||||||
# ------------------------------------------------------------------
|
# ------------------------------------------------------------------
|
||||||
@@ -930,7 +936,7 @@ class NetworkMonitor:
|
|||||||
def run(self) -> None:
|
def run(self) -> None:
|
||||||
logger.info(
|
logger.info(
|
||||||
f'Gandalf monitor started – poll_interval={self.poll_interval}s '
|
f'Gandalf monitor started – poll_interval={self.poll_interval}s '
|
||||||
f'fail_thresh={self.fail_thresh}'
|
f'fail_thresh={self.fail_thresh} unifi_fail_thresh={self.unifi_fail_thresh}'
|
||||||
)
|
)
|
||||||
while True:
|
while True:
|
||||||
try:
|
try:
|
||||||
@@ -942,8 +948,14 @@ class NetworkMonitor:
|
|||||||
# 2. Fetch UniFi devices once — used by both snapshot and alert processing
|
# 2. Fetch UniFi devices once — used by both snapshot and alert processing
|
||||||
unifi_devices = self.unifi.get_devices()
|
unifi_devices = self.unifi.get_devices()
|
||||||
|
|
||||||
# 3. Collect and store snapshot for dashboard
|
# 3a. Ping-only hosts once — shared by snapshot and alert processing
|
||||||
snapshot = self._collect_snapshot(iface_states, unifi_devices)
|
ping_states: Dict[str, bool] = {
|
||||||
|
h['name']: self.pulse.ping(h['ip'])
|
||||||
|
for h in self.cfg.get('monitor', {}).get('ping_hosts', [])
|
||||||
|
}
|
||||||
|
|
||||||
|
# 3b. Collect and store snapshot for dashboard
|
||||||
|
snapshot = self._collect_snapshot(iface_states, unifi_devices, ping_states)
|
||||||
db.set_state('network_snapshot', snapshot)
|
db.set_state('network_snapshot', snapshot)
|
||||||
db.set_state('last_check', _now_utc())
|
db.set_state('last_check', _now_utc())
|
||||||
|
|
||||||
@@ -959,7 +971,7 @@ class NetworkMonitor:
|
|||||||
self._process_interfaces(iface_states, suppressions)
|
self._process_interfaces(iface_states, suppressions)
|
||||||
self._process_unifi(unifi_devices, suppressions)
|
self._process_unifi(unifi_devices, suppressions)
|
||||||
|
|
||||||
self._process_ping_hosts(suppressions)
|
self._process_ping_hosts(suppressions, ping_states)
|
||||||
|
|
||||||
# Housekeeping: deactivate expired suppressions and purge old resolved events
|
# Housekeeping: deactivate expired suppressions and purge old resolved events
|
||||||
db.cleanup_expired_suppressions()
|
db.cleanup_expired_suppressions()
|
||||||
|
|||||||
Reference in New Issue
Block a user