fix: don't ticket the standing Ceph noout flag #28

Merged
jared merged 1 commits from fix/ceph-standing-noout into main 2026-09-25 00:22:04 -04:00
+22 -2
View File
@@ -142,6 +142,11 @@ class SystemHealthMonitor:
'CEPH_TICKET_NODE': None, # Hostname of node designated to create cluster-wide Ceph tickets 'CEPH_TICKET_NODE': None, # Hostname of node designated to create cluster-wide Ceph tickets
'CEPH_USAGE_WARNING': 70, # Ceph cluster usage warning threshold % 'CEPH_USAGE_WARNING': 70, # Ceph cluster usage warning threshold %
'CEPH_USAGE_CRITICAL': 85, # Ceph cluster usage critical threshold % 'CEPH_USAGE_CRITICAL': 85, # Ceph cluster usage critical threshold %
# OSD map flags that are set on purpose and should not raise a HEALTH_WARN ticket.
# noout is a permanent standing flag (large1 has no UPS; see ups-shutdown.sh).
# Only suppressed when these are the ONLY flags set, so e.g. the UPS script's
# nobackfill/norecover/pause still alert.
'CEPH_STANDING_FLAGS': ['noout'],
# Cluster identification for tickets # Cluster identification for tickets
'CLUSTER_NAME': 'proxmox-cluster', # Name used in cluster-wide ticket titles instead of hostname 'CLUSTER_NAME': 'proxmox-cluster', # Name used in cluster-wide ticket titles instead of hostname
# Prometheus metrics settings # Prometheus metrics settings
@@ -3429,6 +3434,16 @@ class SystemHealthMonitor:
'error': str(e) 'error': str(e)
} }
def _only_standing_flags(self, message: str) -> bool:
"""True if an OSDMAP_FLAGS message (e.g. "noout flag(s) set") names only
flags listed in CEPH_STANDING_FLAGS."""
match = re.match(r'^\s*([\w,]+)\s+flag\(s\)\s+set', message or '')
if not match:
return False
flags = {f for f in match.group(1).split(',') if f}
standing = set(self.CONFIG.get('CEPH_STANDING_FLAGS') or [])
return bool(flags) and flags <= standing
def _check_ceph_health(self) -> Dict[str, Any]: def _check_ceph_health(self) -> Dict[str, Any]:
""" """
Check Ceph cluster health if this node is part of a Ceph cluster. Check Ceph cluster health if this node is part of a Ceph cluster.
@@ -3483,16 +3498,21 @@ class SystemHealthMonitor:
f"Ceph cluster HEALTH_ERR: {health_data.get('summary', {}).get('message', 'Unknown error')}" f"Ceph cluster HEALTH_ERR: {health_data.get('summary', {}).get('message', 'Unknown error')}"
) )
elif ceph_health['cluster_health'] == 'HEALTH_WARN': elif ceph_health['cluster_health'] == 'HEALTH_WARN':
if ceph_health['status'] != 'CRITICAL':
ceph_health['status'] = 'WARNING'
# Extract warning messages # Extract warning messages
checks = health_data.get('checks', {}) checks = health_data.get('checks', {})
warned = False
for check_name, check_data in checks.items(): for check_name, check_data in checks.items():
severity = check_data.get('severity', 'HEALTH_WARN') severity = check_data.get('severity', 'HEALTH_WARN')
message = check_data.get('summary', {}).get('message', check_name) message = check_data.get('summary', {}).get('message', check_name)
if check_name == 'OSDMAP_FLAGS' and self._only_standing_flags(message):
logger.debug(f"Ignoring standing Ceph flags: {message}")
continue
warned = True
ceph_health['cluster_wide_issues'].append( ceph_health['cluster_wide_issues'].append(
f"Ceph HEALTH_WARN: {message}" f"Ceph HEALTH_WARN: {message}"
) )
if warned and ceph_health['status'] != 'CRITICAL':
ceph_health['status'] = 'WARNING'
except json.JSONDecodeError as e: except json.JSONDecodeError as e:
logger.warning(f"Failed to parse ceph health JSON: {e}") logger.warning(f"Failed to parse ceph health JSON: {e}")