fix: don't ticket the standing Ceph noout flag
Lint / Python (flake8) (push) Successful in 1m40s
Lint / Notify on failure (push) Skipped
Security / Python Security (bandit) (push) Successful in 1m14s
Test / Python Tests (pytest) (push) Successful in 1m17s
Lint / Python (flake8) (pull_request) Successful in 1m32s
Lint / Notify on failure (pull_request) Skipped
Security / Python Security (bandit) (pull_request) Successful in 1m46s
Test / Python Tests (pytest) (pull_request) Successful in 1m0s

noout is set permanently on purpose (large1 has no UPS; see ups-shutdown.sh),
so hwmonDaemon kept opening/reopening 'Ceph HEALTH_WARN: noout flag(s) set'
(#221040637). Skip the OSDMAP_FLAGS check when the only flags set are listed
in the new CEPH_STANDING_FLAGS config (default ['noout']). Any other flag,
including the UPS script's nobackfill/norecover/pause, still alerts, and a
HEALTH_WARN made only of standing flags no longer marks Ceph as WARNING.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-25 00:18:05 -04:00
co-authored by Claude Opus 5.5
parent 2f6017aa03
commit 2ad911fbf7
+22 -2
View File
@@ -142,6 +142,11 @@ class SystemHealthMonitor:
'CEPH_TICKET_NODE': None, # Hostname of node designated to create cluster-wide Ceph tickets
'CEPH_USAGE_WARNING': 70, # Ceph cluster usage warning threshold %
'CEPH_USAGE_CRITICAL': 85, # Ceph cluster usage critical threshold %
# OSD map flags that are set on purpose and should not raise a HEALTH_WARN ticket.
# noout is a permanent standing flag (large1 has no UPS; see ups-shutdown.sh).
# Only suppressed when these are the ONLY flags set, so e.g. the UPS script's
# nobackfill/norecover/pause still alert.
'CEPH_STANDING_FLAGS': ['noout'],
# Cluster identification for tickets
'CLUSTER_NAME': 'proxmox-cluster', # Name used in cluster-wide ticket titles instead of hostname
# Prometheus metrics settings
@@ -3429,6 +3434,16 @@ class SystemHealthMonitor:
'error': str(e)
}
def _only_standing_flags(self, message: str) -> bool:
"""True if an OSDMAP_FLAGS message (e.g. "noout flag(s) set") names only
flags listed in CEPH_STANDING_FLAGS."""
match = re.match(r'^\s*([\w,]+)\s+flag\(s\)\s+set', message or '')
if not match:
return False
flags = {f for f in match.group(1).split(',') if f}
standing = set(self.CONFIG.get('CEPH_STANDING_FLAGS') or [])
return bool(flags) and flags <= standing
def _check_ceph_health(self) -> Dict[str, Any]:
"""
Check Ceph cluster health if this node is part of a Ceph cluster.
@@ -3483,16 +3498,21 @@ class SystemHealthMonitor:
f"Ceph cluster HEALTH_ERR: {health_data.get('summary', {}).get('message', 'Unknown error')}"
)
elif ceph_health['cluster_health'] == 'HEALTH_WARN':
if ceph_health['status'] != 'CRITICAL':
ceph_health['status'] = 'WARNING'
# Extract warning messages
checks = health_data.get('checks', {})
warned = False
for check_name, check_data in checks.items():
severity = check_data.get('severity', 'HEALTH_WARN')
message = check_data.get('summary', {}).get('message', check_name)
if check_name == 'OSDMAP_FLAGS' and self._only_standing_flags(message):
logger.debug(f"Ignoring standing Ceph flags: {message}")
continue
warned = True
ceph_health['cluster_wide_issues'].append(
f"Ceph HEALTH_WARN: {message}"
)
if warned and ceph_health['status'] != 'CRITICAL':
ceph_health['status'] = 'WARNING'
except json.JSONDecodeError as e:
logger.warning(f"Failed to parse ceph health JSON: {e}")