mirror of
https://github.com/MacRimi/ProxMenux.git
synced 2026-10-09 23:16:41 +00:00
Fix principal cause caps and bind measured recovery to incident policy
This commit is contained in:
@@ -1427,6 +1427,8 @@ class HealthMonitor:
|
||||
'details': f'Sustained for {actual_duration}s above {self.CPU_CRITICAL}%.',
|
||||
'cpu_percent': cpu_percent,
|
||||
'duration': actual_duration,
|
||||
'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
|
||||
'recovery': self.CPU_RECOVERY},
|
||||
},
|
||||
)
|
||||
elif len(warning_samples) >= WARNING_MIN_SAMPLES and len(recovery_samples) < RECOVERY_MIN_SAMPLES:
|
||||
@@ -1446,6 +1448,8 @@ class HealthMonitor:
|
||||
'details': f'Sustained for {actual_duration}s above {self.CPU_WARNING}%.',
|
||||
'cpu_percent': cpu_percent,
|
||||
'duration': actual_duration,
|
||||
'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
|
||||
'recovery': self.CPU_RECOVERY},
|
||||
},
|
||||
)
|
||||
else:
|
||||
@@ -1453,8 +1457,21 @@ class HealthMonitor:
|
||||
reason = None
|
||||
# CPU is normal - auto-resolve any existing CPU errors
|
||||
evidence = None
|
||||
if cpu_percent < self.CPU_RECOVERY and len(recovery_samples) >= RECOVERY_MIN_SAMPLES:
|
||||
evidence = {'check': 'cpu_usage', 'checked_at': current_time}
|
||||
# Presentation proof is stricter than operational hysteresis:
|
||||
# a supported warning can be below the fixed recovery cutoff.
|
||||
from health_recovery import _finite_number
|
||||
criterion = min(self.CPU_WARNING, self.CPU_RECOVERY)
|
||||
normal_samples = [entry for entry in self.state_history[state_key]
|
||||
if _finite_number(entry['value']) and 0 <= entry['value'] < criterion
|
||||
and _finite_number(entry['time'])
|
||||
and 0 <= current_time - entry['time'] <= self.CPU_RECOVERY_DURATION]
|
||||
if (_finite_number(cpu_percent) and 0 <= cpu_percent < criterion
|
||||
and len(normal_samples) >= RECOVERY_MIN_SAMPLES):
|
||||
evidence = {'check': 'cpu_usage', 'checked_at': current_time,
|
||||
'value': cpu_percent, 'normal_samples': len(normal_samples),
|
||||
'max_sample': max(entry['value'] for entry in normal_samples),
|
||||
'policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
|
||||
'recovery': self.CPU_RECOVERY}}
|
||||
health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal',
|
||||
check_evidence=evidence)
|
||||
|
||||
@@ -3904,6 +3921,7 @@ class HealthMonitor:
|
||||
|
||||
failed_services = []
|
||||
service_details = {}
|
||||
active_evidence = {}
|
||||
|
||||
for service in services_to_check:
|
||||
try:
|
||||
@@ -3918,6 +3936,10 @@ class HealthMonitor:
|
||||
if result.returncode != 0 or status != 'active':
|
||||
failed_services.append(service)
|
||||
service_details[service] = status or 'inactive'
|
||||
else:
|
||||
active_evidence[service] = {'check': f'pve_service_{service}',
|
||||
'checked_at': time.time(), 'service': service, 'state': status,
|
||||
'returncode': result.returncode}
|
||||
except Exception:
|
||||
failed_services.append(service)
|
||||
service_details[service] = 'error'
|
||||
@@ -3932,7 +3954,7 @@ class HealthMonitor:
|
||||
if svc not in failed_services:
|
||||
error_key = f'pve_service_{svc}'
|
||||
if health_persistence.is_error_active(error_key):
|
||||
health_persistence.clear_error(error_key)
|
||||
health_persistence.clear_error(error_key, check_evidence=active_evidence.get(svc))
|
||||
|
||||
# Build checks dict with status per service
|
||||
checks = {}
|
||||
|
||||
@@ -600,7 +600,7 @@ class HealthPersistence:
|
||||
|
||||
cursor.execute('''
|
||||
SELECT id, acknowledged, resolved_at, category, severity, first_seen,
|
||||
notification_sent, suppression_hours, acknowledged_at
|
||||
notification_sent, suppression_hours, acknowledged_at, details
|
||||
FROM errors WHERE error_key = ?
|
||||
''', (error_key,))
|
||||
existing = cursor.fetchone()
|
||||
@@ -609,7 +609,7 @@ class HealthPersistence:
|
||||
|
||||
if existing:
|
||||
(err_id, ack, resolved_at, old_cat, old_severity, first_seen,
|
||||
notif_sent, stored_suppression, acknowledged_at) = existing
|
||||
notif_sent, stored_suppression, acknowledged_at, old_details_json) = existing
|
||||
|
||||
if ack == 1:
|
||||
# SAFETY OVERRIDE: Critical CPU temperature ALWAYS re-triggers
|
||||
@@ -680,6 +680,18 @@ class HealthPersistence:
|
||||
conn.commit()
|
||||
return event_info
|
||||
|
||||
# Original CPU policy is immutable for this row's incident.
|
||||
# Never upgrade a legacy row from later/current settings.
|
||||
if error_key == 'cpu_usage':
|
||||
try:
|
||||
old_details = json.loads(old_details_json or '{}')
|
||||
except (ValueError, TypeError):
|
||||
old_details = {}
|
||||
details = dict(details) if isinstance(details, dict) else {}
|
||||
details.pop('cpu_policy', None)
|
||||
if isinstance(old_details, dict) and 'cpu_policy' in old_details:
|
||||
details['cpu_policy'] = old_details['cpu_policy']
|
||||
details_json = json.dumps(details)
|
||||
# Not acknowledged - update existing active error
|
||||
cursor.execute('''
|
||||
UPDATE errors
|
||||
@@ -776,7 +788,7 @@ class HealthPersistence:
|
||||
# was created — otherwise "Storage 'Tuxis' unavailable"
|
||||
# comes back as "Resolved - Storage" with no identity.
|
||||
cursor.execute(
|
||||
'SELECT details FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1',
|
||||
'SELECT details, id, first_seen, resolved_at FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1',
|
||||
(error_key,),
|
||||
)
|
||||
row = cursor.fetchone()
|
||||
@@ -786,11 +798,18 @@ class HealthPersistence:
|
||||
stored_details = json.loads(row[0])
|
||||
except Exception:
|
||||
stored_details = None
|
||||
# Legacy/mismatched original policy cannot certify normality.
|
||||
if error_key == 'cpu_usage' and (not isinstance(stored_details, dict)
|
||||
or not isinstance(check_evidence, dict)
|
||||
or check_evidence.get('policy') != stored_details.get('cpu_policy')
|
||||
or not stored_details.get('cpu_policy')):
|
||||
check_evidence = None
|
||||
self._record_event(cursor, 'resolved', error_key, {
|
||||
'reason': reason,
|
||||
# Only explicit current-check callers attach this proof.
|
||||
# Generic resolve/cleanup remains neutral.
|
||||
'check_evidence': check_evidence,
|
||||
'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': row[3]},
|
||||
'entity': self._entity_from_details(stored_details),
|
||||
'details': stored_details or {},
|
||||
})
|
||||
@@ -800,41 +819,40 @@ class HealthPersistence:
|
||||
def get_recovery_evidence(self, error_key: str, first_seen: str):
|
||||
"""Return fresh same-incident native check proof, never absence of errors.
|
||||
|
||||
Initially only the host CPU check has a stable condition identity. Other
|
||||
Host CPU and exact per-service active checks carry provenance. Other
|
||||
checks, generic clears, excluded/deleted records and legacy events stay
|
||||
neutral until they have equivalent per-condition provenance.
|
||||
"""
|
||||
if error_key != 'cpu_usage' or not first_seen:
|
||||
if not isinstance(error_key, str) or not first_seen or not (
|
||||
error_key == 'cpu_usage' or error_key.startswith('pve_service_')):
|
||||
return None
|
||||
try:
|
||||
with self._db_connection() as conn:
|
||||
# One SQLite statement is one consistent row/ack/closure snapshot.
|
||||
# Latest closure by event id, never search past a generic clear.
|
||||
with self._db_lock, self._db_connection() as conn:
|
||||
row = conn.execute('''
|
||||
SELECT first_seen, last_seen, resolved_at, acknowledged
|
||||
FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1
|
||||
SELECT e.first_seen, e.last_seen, e.resolved_at, e.acknowledged,
|
||||
e.id, v.timestamp, v.data
|
||||
FROM errors e JOIN events v ON v.id = (
|
||||
SELECT id FROM events WHERE error_key = e.error_key
|
||||
AND event_type IN ('resolved', 'cleared') ORDER BY id DESC LIMIT 1
|
||||
) WHERE e.error_key = ?
|
||||
''', (error_key,)).fetchone()
|
||||
if not row or row[0] != first_seen or not row[2] or row[3]:
|
||||
return None
|
||||
event = conn.execute('''
|
||||
SELECT timestamp, data FROM events
|
||||
WHERE error_key = ? AND event_type = 'resolved'
|
||||
ORDER BY id DESC LIMIT 1
|
||||
''', (error_key,)).fetchone()
|
||||
if not event:
|
||||
if not row or row[0] != first_seen or not row[2] or row[3]:
|
||||
return None
|
||||
proof = json.loads(event[1]).get('check_evidence')
|
||||
if not isinstance(proof, dict) or proof.get('check') != error_key:
|
||||
event_data = json.loads(row[6])
|
||||
if event_data.get('incident') != {'id': row[4], 'first_seen': row[0], 'resolved_at': row[2]}:
|
||||
return None
|
||||
checked = proof.get('checked_at')
|
||||
if not isinstance(checked, (int, float)) or isinstance(checked, bool):
|
||||
proof = event_data.get('check_evidence')
|
||||
from health_recovery import valid_check_evidence
|
||||
if not valid_check_evidence(error_key, proof, now=datetime.now().timestamp()):
|
||||
return None
|
||||
checked = float(checked)
|
||||
# Reuse the collector's existing two-hour freshness boundary.
|
||||
now = datetime.now().timestamp()
|
||||
if not 0 <= now - checked <= 7200:
|
||||
if error_key == 'cpu_usage' and proof.get('policy') != event_data.get('details', {}).get('cpu_policy'):
|
||||
return None
|
||||
checked = float(proof['checked_at'])
|
||||
last_seen = datetime.fromisoformat(row[1]).timestamp()
|
||||
resolved = datetime.fromisoformat(row[2]).timestamp()
|
||||
recorded = datetime.fromisoformat(event[0]).timestamp()
|
||||
recorded = datetime.fromisoformat(row[5]).timestamp()
|
||||
if not last_seen <= checked <= resolved <= recorded:
|
||||
return None
|
||||
return proof
|
||||
@@ -906,7 +924,7 @@ class HealthPersistence:
|
||||
|
||||
return False
|
||||
|
||||
def clear_error(self, error_key: str):
|
||||
def clear_error(self, error_key: str, *, check_evidence=None):
|
||||
"""
|
||||
Remove/resolve a specific error immediately.
|
||||
Used when the condition that caused the error no longer exists
|
||||
@@ -928,7 +946,7 @@ class HealthPersistence:
|
||||
|
||||
# Check if this error was acknowledged (dismissed)
|
||||
cursor.execute('''
|
||||
SELECT acknowledged FROM errors WHERE error_key = ?
|
||||
SELECT acknowledged, id, first_seen FROM errors WHERE error_key = ?
|
||||
''', (error_key,))
|
||||
row = cursor.fetchone()
|
||||
|
||||
@@ -947,7 +965,10 @@ class HealthPersistence:
|
||||
''', (now, error_key))
|
||||
|
||||
if cursor.rowcount > 0:
|
||||
self._record_event(cursor, 'cleared', error_key, {'reason': 'condition_resolved'})
|
||||
self._record_event(cursor, 'cleared', error_key, {
|
||||
'reason': 'condition_resolved', 'check_evidence': check_evidence,
|
||||
'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': now},
|
||||
})
|
||||
|
||||
conn.commit()
|
||||
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Bounded recovery metadata admission, without importing monitor singletons.
|
||||
|
||||
Native persistence additionally binds the exact incident and closure. Manual
|
||||
notifications remain authenticated caller assertions: shape validation cannot
|
||||
establish that an asserted measurement actually happened.
|
||||
"""
|
||||
import math
|
||||
import time
|
||||
from typing import TypeGuard
|
||||
|
||||
|
||||
def _finite_number(value) -> TypeGuard[int | float]:
|
||||
try:
|
||||
return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value)
|
||||
except (OverflowError, ValueError, TypeError):
|
||||
return False
|
||||
|
||||
|
||||
def valid_check_evidence(error_key, proof, *, now=None):
|
||||
"""Validate a supported measurement contract, not its external authenticity."""
|
||||
if not isinstance(proof, dict) or proof.get('check') != error_key:
|
||||
return False
|
||||
checked = proof.get('checked_at')
|
||||
now = time.time() if now is None else now
|
||||
if not _finite_number(checked) or not _finite_number(now) or not 0 <= now-checked <= 7200:
|
||||
return False
|
||||
if error_key == 'cpu_usage':
|
||||
policy = proof.get('policy')
|
||||
if not isinstance(policy, dict) or set(policy) != {'warning', 'critical', 'recovery'}:
|
||||
return False
|
||||
if not all(_finite_number(v) and 1 <= v <= 100 for v in policy.values()):
|
||||
return False
|
||||
if policy['warning'] > policy['critical']:
|
||||
return False
|
||||
value, maximum, count = proof.get('value'), proof.get('max_sample'), proof.get('normal_samples')
|
||||
return (_finite_number(value) and _finite_number(maximum)
|
||||
and 0 <= value <= maximum < min(policy['warning'], policy['recovery'])
|
||||
and isinstance(count, int) and not isinstance(count, bool) and count >= 10)
|
||||
if isinstance(error_key, str) and error_key.startswith('pve_service_'):
|
||||
service = error_key[len('pve_service_'):]
|
||||
return (bool(service) and proof.get('service') == service
|
||||
and proof.get('state') == 'active'
|
||||
and type(proof.get('returncode')) is int and proof['returncode'] == 0)
|
||||
return False
|
||||
|
||||
|
||||
def presents_recovery(data):
|
||||
"""One presentation predicate shared by template, icon and email badge."""
|
||||
return (isinstance(data, dict) and data.get('recovery_outcome') == 'resolved'
|
||||
and data.get('is_recovery') is True
|
||||
and valid_check_evidence(data.get('error_key'), data.get('check_evidence')))
|
||||
@@ -1038,10 +1038,8 @@ class EmailChannel(NotificationChannel):
|
||||
# Determine group for section header
|
||||
event_type = data.get('_event_type', '')
|
||||
if event_type == 'error_resolved':
|
||||
if (data.get('recovery_outcome') == 'resolved'
|
||||
and data.get('is_recovery') is True
|
||||
and isinstance(data.get('check_evidence'), dict)
|
||||
and data['check_evidence'].get('check') == 'cpu_usage'):
|
||||
from health_recovery import presents_recovery
|
||||
if presents_recovery(data):
|
||||
sev.update(self._SEV_STYLE['OK'])
|
||||
sev['label'] = _runtime_notification_text('healthRecovery.status', data)
|
||||
else:
|
||||
@@ -1070,8 +1068,11 @@ class EmailChannel(NotificationChannel):
|
||||
or backup_email or data.get('_restore_summary') or data.get('_backup_summary'))
|
||||
temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else ''
|
||||
temp_table_layout = 'table-layout:fixed;' if wrap_body else ''
|
||||
backup_title_wrap = temp_cell_wrap if backup_email else ''
|
||||
backup_metadata_layout = 'table-layout:fixed;' if backup_email else ''
|
||||
# Recovery exposes the same literal host context as backup notices.
|
||||
# Keep wrapping event-scoped; unrelated mail remains byte-identical.
|
||||
context_email = backup_email or event_type == 'error_resolved'
|
||||
backup_title_wrap = temp_cell_wrap if context_email else ''
|
||||
backup_metadata_layout = 'table-layout:fixed;' if context_email else ''
|
||||
section_label = _runtime_text(f'email.groups.{group}', data)
|
||||
report_label = _runtime_text('email.report', data, group=section_label)
|
||||
host_label = _runtime_text('email.host', data)
|
||||
|
||||
@@ -2070,10 +2070,8 @@ def render_template(event_type: str, data: Dict[str, Any],
|
||||
return ''
|
||||
|
||||
safe_vars = _SafeDict(variables)
|
||||
if (event_type == 'error_resolved' and data.get('recovery_outcome') == 'resolved'
|
||||
and data.get('is_recovery') is True
|
||||
and isinstance(data.get('check_evidence'), dict)
|
||||
and data['check_evidence'].get('check') == 'cpu_usage'):
|
||||
from health_recovery import presents_recovery
|
||||
if event_type == 'error_resolved' and presents_recovery(data):
|
||||
safe_vars['_health_title'] = runtime_message('healthRecovery.title', language, **variables)
|
||||
safe_vars['_health_body'] = runtime_message('healthRecovery.body', language, **variables)
|
||||
template['title'] = '{_health_title}'
|
||||
@@ -2089,13 +2087,14 @@ def render_template(event_type: str, data: Dict[str, Any],
|
||||
# parse the table/logs and format a rich body instead of the sparse template.
|
||||
pve_message = data.get('pve_message', '')
|
||||
backup_diagnostics = []
|
||||
principal_cause = None
|
||||
|
||||
def bounded_backup_diagnostics(lines):
|
||||
def bounded_backup_diagnostics(lines, principal_cause=None):
|
||||
# 1024 chars matches the repository's small-channel message convention;
|
||||
# 8 lines keeps repeated producer warnings readable. Inventory/title
|
||||
# size is separate: this is not a one-Telegram-message guarantee.
|
||||
unique = list(dict.fromkeys(line for line in lines if line.strip()))
|
||||
principal = next((line for line in unique if re.search(r'\b(?:ERROR:|TASK ERROR:)', line, re.IGNORECASE)), None)
|
||||
principal = principal_cause or next((line for line in unique if re.search(r'\b(?:ERROR:|TASK ERROR:)', line, re.IGNORECASE)), None)
|
||||
if principal:
|
||||
unique.remove(principal)
|
||||
unique.insert(0, principal)
|
||||
@@ -2189,16 +2188,18 @@ def render_template(event_type: str, data: Dict[str, Any],
|
||||
source_subject = cause.group(1).strip() if cause else ''
|
||||
if source_subject.lower() == 'multiple problems':
|
||||
source_subject = ''
|
||||
cause_in_diagnostics = any(
|
||||
line.strip() == source_subject or
|
||||
re.split(r'\b(?:TASK ERROR:|ERROR:)\s*', line, maxsplit=1, flags=re.IGNORECASE)[-1].strip() == source_subject
|
||||
for line in backup_diagnostics)
|
||||
if source_subject and source_subject not in body_text and not cause_in_diagnostics:
|
||||
# A unique job/setup cause must survive a warning-heavy report.
|
||||
backup_diagnostics.insert(0, source_subject)
|
||||
if source_subject and source_subject not in body_text:
|
||||
# Reserve the subject-equivalent diagnostic BEFORE the cap. Finding
|
||||
# it in uncapped logs is not enough: that late line could be omitted.
|
||||
principal_cause = next((line for line in backup_diagnostics
|
||||
if line.strip() == source_subject or
|
||||
re.split(r'\b(?:TASK ERROR:|ERROR:)\s*', line, maxsplit=1, flags=re.IGNORECASE)[-1].strip() == source_subject), None)
|
||||
if not principal_cause:
|
||||
principal_cause = source_subject
|
||||
backup_diagnostics.insert(0, source_subject)
|
||||
|
||||
if backup_diagnostics:
|
||||
body_text += '\n' + bounded_backup_diagnostics(backup_diagnostics)
|
||||
body_text += '\n' + bounded_backup_diagnostics(backup_diagnostics, principal_cause)
|
||||
|
||||
# Clean up: collapse runs of 3+ blank lines into 1, remove trailing whitespace
|
||||
import re as _re
|
||||
@@ -2531,10 +2532,8 @@ def enrich_with_emojis(event_type: str, title: str, body: str,
|
||||
severity = data.get('severity', 'INFO')
|
||||
|
||||
icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '')
|
||||
if (event_type == 'error_resolved' and data.get('recovery_outcome') == 'resolved'
|
||||
and data.get('is_recovery') is True
|
||||
and isinstance(data.get('check_evidence'), dict)
|
||||
and data['check_evidence'].get('check') == 'cpu_usage'):
|
||||
from health_recovery import presents_recovery
|
||||
if event_type == 'error_resolved' and presents_recovery(data):
|
||||
icon = '✅'
|
||||
if event_type == 'backup_complete':
|
||||
icon = {
|
||||
|
||||
Reference in New Issue
Block a user