From 2ad911fbf7e6e7ed84662b7cf010ff1b258fa044 Mon Sep 17 00:00:00 2001 From: Jared Vititoe Date: Fri, 25 Sep 2026 00:18:05 -0400 Subject: [PATCH] fix: don't ticket the standing Ceph noout flag noout is set permanently on purpose (large1 has no UPS; see ups-shutdown.sh), so hwmonDaemon kept opening/reopening 'Ceph HEALTH_WARN: noout flag(s) set' (#221040637). Skip the OSDMAP_FLAGS check when the only flags set are listed in the new CEPH_STANDING_FLAGS config (default ['noout']). Any other flag, including the UPS script's nobackfill/norecover/pause, still alerts, and a HEALTH_WARN made only of standing flags no longer marks Ceph as WARNING. Co-Authored-By: Claude Opus 5.5 --- hwmonDaemon.py | 24 ++++++++++++++++++++++-- 1 file changed, 22 insertions(+), 2 deletions(-) diff --git a/hwmonDaemon.py b/hwmonDaemon.py index f47a229..afc238d 100644 --- a/hwmonDaemon.py +++ b/hwmonDaemon.py @@ -142,6 +142,11 @@ class SystemHealthMonitor: 'CEPH_TICKET_NODE': None, # Hostname of node designated to create cluster-wide Ceph tickets 'CEPH_USAGE_WARNING': 70, # Ceph cluster usage warning threshold % 'CEPH_USAGE_CRITICAL': 85, # Ceph cluster usage critical threshold % + # OSD map flags that are set on purpose and should not raise a HEALTH_WARN ticket. + # noout is a permanent standing flag (large1 has no UPS; see ups-shutdown.sh). + # Only suppressed when these are the ONLY flags set, so e.g. the UPS script's + # nobackfill/norecover/pause still alert. + 'CEPH_STANDING_FLAGS': ['noout'], # Cluster identification for tickets 'CLUSTER_NAME': 'proxmox-cluster', # Name used in cluster-wide ticket titles instead of hostname # Prometheus metrics settings @@ -3429,6 +3434,16 @@ class SystemHealthMonitor: 'error': str(e) } + def _only_standing_flags(self, message: str) -> bool: + """True if an OSDMAP_FLAGS message (e.g. "noout flag(s) set") names only + flags listed in CEPH_STANDING_FLAGS.""" + match = re.match(r'^\s*([\w,]+)\s+flag\(s\)\s+set', message or '') + if not match: + return False + flags = {f for f in match.group(1).split(',') if f} + standing = set(self.CONFIG.get('CEPH_STANDING_FLAGS') or []) + return bool(flags) and flags <= standing + def _check_ceph_health(self) -> Dict[str, Any]: """ Check Ceph cluster health if this node is part of a Ceph cluster. @@ -3483,16 +3498,21 @@ class SystemHealthMonitor: f"Ceph cluster HEALTH_ERR: {health_data.get('summary', {}).get('message', 'Unknown error')}" ) elif ceph_health['cluster_health'] == 'HEALTH_WARN': - if ceph_health['status'] != 'CRITICAL': - ceph_health['status'] = 'WARNING' # Extract warning messages checks = health_data.get('checks', {}) + warned = False for check_name, check_data in checks.items(): severity = check_data.get('severity', 'HEALTH_WARN') message = check_data.get('summary', {}).get('message', check_name) + if check_name == 'OSDMAP_FLAGS' and self._only_standing_flags(message): + logger.debug(f"Ignoring standing Ceph flags: {message}") + continue + warned = True ceph_health['cluster_wide_issues'].append( f"Ceph HEALTH_WARN: {message}" ) + if warned and ceph_health['status'] != 'CRITICAL': + ceph_health['status'] = 'WARNING' except json.JSONDecodeError as e: logger.warning(f"Failed to parse ceph health JSON: {e}")