fix: don't ticket the standing Ceph noout flag
Lint / Python (flake8) (push) Successful in 1m40s
Lint / Notify on failure (push) Skipped
Security / Python Security (bandit) (push) Successful in 1m14s
Test / Python Tests (pytest) (push) Successful in 1m17s
Lint / Python (flake8) (pull_request) Successful in 1m32s
Lint / Notify on failure (pull_request) Skipped
Security / Python Security (bandit) (pull_request) Successful in 1m46s
Test / Python Tests (pytest) (pull_request) Successful in 1m0s
Lint / Python (flake8) (push) Successful in 1m40s
Lint / Notify on failure (push) Skipped
Security / Python Security (bandit) (push) Successful in 1m14s
Test / Python Tests (pytest) (push) Successful in 1m17s
Lint / Python (flake8) (pull_request) Successful in 1m32s
Lint / Notify on failure (pull_request) Skipped
Security / Python Security (bandit) (pull_request) Successful in 1m46s
Test / Python Tests (pytest) (pull_request) Successful in 1m0s
noout is set permanently on purpose (large1 has no UPS; see ups-shutdown.sh), so hwmonDaemon kept opening/reopening 'Ceph HEALTH_WARN: noout flag(s) set' (#221040637). Skip the OSDMAP_FLAGS check when the only flags set are listed in the new CEPH_STANDING_FLAGS config (default ['noout']). Any other flag, including the UPS script's nobackfill/norecover/pause, still alerts, and a HEALTH_WARN made only of standing flags no longer marks Ceph as WARNING. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
+22
-2
@@ -142,6 +142,11 @@ class SystemHealthMonitor:
|
|||||||
'CEPH_TICKET_NODE': None, # Hostname of node designated to create cluster-wide Ceph tickets
|
'CEPH_TICKET_NODE': None, # Hostname of node designated to create cluster-wide Ceph tickets
|
||||||
'CEPH_USAGE_WARNING': 70, # Ceph cluster usage warning threshold %
|
'CEPH_USAGE_WARNING': 70, # Ceph cluster usage warning threshold %
|
||||||
'CEPH_USAGE_CRITICAL': 85, # Ceph cluster usage critical threshold %
|
'CEPH_USAGE_CRITICAL': 85, # Ceph cluster usage critical threshold %
|
||||||
|
# OSD map flags that are set on purpose and should not raise a HEALTH_WARN ticket.
|
||||||
|
# noout is a permanent standing flag (large1 has no UPS; see ups-shutdown.sh).
|
||||||
|
# Only suppressed when these are the ONLY flags set, so e.g. the UPS script's
|
||||||
|
# nobackfill/norecover/pause still alert.
|
||||||
|
'CEPH_STANDING_FLAGS': ['noout'],
|
||||||
# Cluster identification for tickets
|
# Cluster identification for tickets
|
||||||
'CLUSTER_NAME': 'proxmox-cluster', # Name used in cluster-wide ticket titles instead of hostname
|
'CLUSTER_NAME': 'proxmox-cluster', # Name used in cluster-wide ticket titles instead of hostname
|
||||||
# Prometheus metrics settings
|
# Prometheus metrics settings
|
||||||
@@ -3429,6 +3434,16 @@ class SystemHealthMonitor:
|
|||||||
'error': str(e)
|
'error': str(e)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
def _only_standing_flags(self, message: str) -> bool:
|
||||||
|
"""True if an OSDMAP_FLAGS message (e.g. "noout flag(s) set") names only
|
||||||
|
flags listed in CEPH_STANDING_FLAGS."""
|
||||||
|
match = re.match(r'^\s*([\w,]+)\s+flag\(s\)\s+set', message or '')
|
||||||
|
if not match:
|
||||||
|
return False
|
||||||
|
flags = {f for f in match.group(1).split(',') if f}
|
||||||
|
standing = set(self.CONFIG.get('CEPH_STANDING_FLAGS') or [])
|
||||||
|
return bool(flags) and flags <= standing
|
||||||
|
|
||||||
def _check_ceph_health(self) -> Dict[str, Any]:
|
def _check_ceph_health(self) -> Dict[str, Any]:
|
||||||
"""
|
"""
|
||||||
Check Ceph cluster health if this node is part of a Ceph cluster.
|
Check Ceph cluster health if this node is part of a Ceph cluster.
|
||||||
@@ -3483,16 +3498,21 @@ class SystemHealthMonitor:
|
|||||||
f"Ceph cluster HEALTH_ERR: {health_data.get('summary', {}).get('message', 'Unknown error')}"
|
f"Ceph cluster HEALTH_ERR: {health_data.get('summary', {}).get('message', 'Unknown error')}"
|
||||||
)
|
)
|
||||||
elif ceph_health['cluster_health'] == 'HEALTH_WARN':
|
elif ceph_health['cluster_health'] == 'HEALTH_WARN':
|
||||||
if ceph_health['status'] != 'CRITICAL':
|
|
||||||
ceph_health['status'] = 'WARNING'
|
|
||||||
# Extract warning messages
|
# Extract warning messages
|
||||||
checks = health_data.get('checks', {})
|
checks = health_data.get('checks', {})
|
||||||
|
warned = False
|
||||||
for check_name, check_data in checks.items():
|
for check_name, check_data in checks.items():
|
||||||
severity = check_data.get('severity', 'HEALTH_WARN')
|
severity = check_data.get('severity', 'HEALTH_WARN')
|
||||||
message = check_data.get('summary', {}).get('message', check_name)
|
message = check_data.get('summary', {}).get('message', check_name)
|
||||||
|
if check_name == 'OSDMAP_FLAGS' and self._only_standing_flags(message):
|
||||||
|
logger.debug(f"Ignoring standing Ceph flags: {message}")
|
||||||
|
continue
|
||||||
|
warned = True
|
||||||
ceph_health['cluster_wide_issues'].append(
|
ceph_health['cluster_wide_issues'].append(
|
||||||
f"Ceph HEALTH_WARN: {message}"
|
f"Ceph HEALTH_WARN: {message}"
|
||||||
)
|
)
|
||||||
|
if warned and ceph_health['status'] != 'CRITICAL':
|
||||||
|
ceph_health['status'] = 'WARNING'
|
||||||
except json.JSONDecodeError as e:
|
except json.JSONDecodeError as e:
|
||||||
logger.warning(f"Failed to parse ceph health JSON: {e}")
|
logger.warning(f"Failed to parse ceph health JSON: {e}")
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user