fix: don't ticket the standing Ceph noout flag
Lint / Python (flake8) (push) Successful in 1m40s
Lint / Notify on failure (push) Skipped
Security / Python Security (bandit) (push) Successful in 1m14s
Test / Python Tests (pytest) (push) Successful in 1m17s
Lint / Python (flake8) (pull_request) Successful in 1m32s
Lint / Notify on failure (pull_request) Skipped
Security / Python Security (bandit) (pull_request) Successful in 1m46s
Test / Python Tests (pytest) (pull_request) Successful in 1m0s
Lint / Python (flake8) (push) Successful in 1m40s
Lint / Notify on failure (push) Skipped
Security / Python Security (bandit) (push) Successful in 1m14s
Test / Python Tests (pytest) (push) Successful in 1m17s
Lint / Python (flake8) (pull_request) Successful in 1m32s
Lint / Notify on failure (pull_request) Skipped
Security / Python Security (bandit) (pull_request) Successful in 1m46s
Test / Python Tests (pytest) (pull_request) Successful in 1m0s
noout is set permanently on purpose (large1 has no UPS; see ups-shutdown.sh), so hwmonDaemon kept opening/reopening 'Ceph HEALTH_WARN: noout flag(s) set' (#221040637). Skip the OSDMAP_FLAGS check when the only flags set are listed in the new CEPH_STANDING_FLAGS config (default ['noout']). Any other flag, including the UPS script's nobackfill/norecover/pause, still alerts, and a HEALTH_WARN made only of standing flags no longer marks Ceph as WARNING. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
+22
-2
@@ -142,6 +142,11 @@ class SystemHealthMonitor:
|
||||
'CEPH_TICKET_NODE': None, # Hostname of node designated to create cluster-wide Ceph tickets
|
||||
'CEPH_USAGE_WARNING': 70, # Ceph cluster usage warning threshold %
|
||||
'CEPH_USAGE_CRITICAL': 85, # Ceph cluster usage critical threshold %
|
||||
# OSD map flags that are set on purpose and should not raise a HEALTH_WARN ticket.
|
||||
# noout is a permanent standing flag (large1 has no UPS; see ups-shutdown.sh).
|
||||
# Only suppressed when these are the ONLY flags set, so e.g. the UPS script's
|
||||
# nobackfill/norecover/pause still alert.
|
||||
'CEPH_STANDING_FLAGS': ['noout'],
|
||||
# Cluster identification for tickets
|
||||
'CLUSTER_NAME': 'proxmox-cluster', # Name used in cluster-wide ticket titles instead of hostname
|
||||
# Prometheus metrics settings
|
||||
@@ -3429,6 +3434,16 @@ class SystemHealthMonitor:
|
||||
'error': str(e)
|
||||
}
|
||||
|
||||
def _only_standing_flags(self, message: str) -> bool:
|
||||
"""True if an OSDMAP_FLAGS message (e.g. "noout flag(s) set") names only
|
||||
flags listed in CEPH_STANDING_FLAGS."""
|
||||
match = re.match(r'^\s*([\w,]+)\s+flag\(s\)\s+set', message or '')
|
||||
if not match:
|
||||
return False
|
||||
flags = {f for f in match.group(1).split(',') if f}
|
||||
standing = set(self.CONFIG.get('CEPH_STANDING_FLAGS') or [])
|
||||
return bool(flags) and flags <= standing
|
||||
|
||||
def _check_ceph_health(self) -> Dict[str, Any]:
|
||||
"""
|
||||
Check Ceph cluster health if this node is part of a Ceph cluster.
|
||||
@@ -3483,16 +3498,21 @@ class SystemHealthMonitor:
|
||||
f"Ceph cluster HEALTH_ERR: {health_data.get('summary', {}).get('message', 'Unknown error')}"
|
||||
)
|
||||
elif ceph_health['cluster_health'] == 'HEALTH_WARN':
|
||||
if ceph_health['status'] != 'CRITICAL':
|
||||
ceph_health['status'] = 'WARNING'
|
||||
# Extract warning messages
|
||||
checks = health_data.get('checks', {})
|
||||
warned = False
|
||||
for check_name, check_data in checks.items():
|
||||
severity = check_data.get('severity', 'HEALTH_WARN')
|
||||
message = check_data.get('summary', {}).get('message', check_name)
|
||||
if check_name == 'OSDMAP_FLAGS' and self._only_standing_flags(message):
|
||||
logger.debug(f"Ignoring standing Ceph flags: {message}")
|
||||
continue
|
||||
warned = True
|
||||
ceph_health['cluster_wide_issues'].append(
|
||||
f"Ceph HEALTH_WARN: {message}"
|
||||
)
|
||||
if warned and ceph_health['status'] != 'CRITICAL':
|
||||
ceph_health['status'] = 'WARNING'
|
||||
except json.JSONDecodeError as e:
|
||||
logger.warning(f"Failed to parse ceph health JSON: {e}")
|
||||
|
||||
|
||||
Reference in New Issue
Block a user