Fix principal cause caps and bind measured recovery to incident policy

This commit is contained in:
martino
2026-10-01 08:38:32 +02:00
parent 8e396f957f
commit 3be15211b1
9 changed files with 537 additions and 140 deletions
+25 -3
View File
@@ -1427,6 +1427,8 @@ class HealthMonitor:
'details': f'Sustained for {actual_duration}s above {self.CPU_CRITICAL}%.',
'cpu_percent': cpu_percent,
'duration': actual_duration,
'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
'recovery': self.CPU_RECOVERY},
},
)
elif len(warning_samples) >= WARNING_MIN_SAMPLES and len(recovery_samples) < RECOVERY_MIN_SAMPLES:
@@ -1446,6 +1448,8 @@ class HealthMonitor:
'details': f'Sustained for {actual_duration}s above {self.CPU_WARNING}%.',
'cpu_percent': cpu_percent,
'duration': actual_duration,
'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
'recovery': self.CPU_RECOVERY},
},
)
else:
@@ -1453,8 +1457,21 @@ class HealthMonitor:
reason = None
# CPU is normal - auto-resolve any existing CPU errors
evidence = None
if cpu_percent < self.CPU_RECOVERY and len(recovery_samples) >= RECOVERY_MIN_SAMPLES:
evidence = {'check': 'cpu_usage', 'checked_at': current_time}
# Presentation proof is stricter than operational hysteresis:
# a supported warning can be below the fixed recovery cutoff.
from health_recovery import _finite_number
criterion = min(self.CPU_WARNING, self.CPU_RECOVERY)
normal_samples = [entry for entry in self.state_history[state_key]
if _finite_number(entry['value']) and 0 <= entry['value'] < criterion
and _finite_number(entry['time'])
and 0 <= current_time - entry['time'] <= self.CPU_RECOVERY_DURATION]
if (_finite_number(cpu_percent) and 0 <= cpu_percent < criterion
and len(normal_samples) >= RECOVERY_MIN_SAMPLES):
evidence = {'check': 'cpu_usage', 'checked_at': current_time,
'value': cpu_percent, 'normal_samples': len(normal_samples),
'max_sample': max(entry['value'] for entry in normal_samples),
'policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
'recovery': self.CPU_RECOVERY}}
health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal',
check_evidence=evidence)
@@ -3904,6 +3921,7 @@ class HealthMonitor:
failed_services = []
service_details = {}
active_evidence = {}
for service in services_to_check:
try:
@@ -3918,6 +3936,10 @@ class HealthMonitor:
if result.returncode != 0 or status != 'active':
failed_services.append(service)
service_details[service] = status or 'inactive'
else:
active_evidence[service] = {'check': f'pve_service_{service}',
'checked_at': time.time(), 'service': service, 'state': status,
'returncode': result.returncode}
except Exception:
failed_services.append(service)
service_details[service] = 'error'
@@ -3932,7 +3954,7 @@ class HealthMonitor:
if svc not in failed_services:
error_key = f'pve_service_{svc}'
if health_persistence.is_error_active(error_key):
health_persistence.clear_error(error_key)
health_persistence.clear_error(error_key, check_evidence=active_evidence.get(svc))
# Build checks dict with status per service
checks = {}
+49 -28
View File
@@ -600,7 +600,7 @@ class HealthPersistence:
cursor.execute('''
SELECT id, acknowledged, resolved_at, category, severity, first_seen,
notification_sent, suppression_hours, acknowledged_at
notification_sent, suppression_hours, acknowledged_at, details
FROM errors WHERE error_key = ?
''', (error_key,))
existing = cursor.fetchone()
@@ -609,7 +609,7 @@ class HealthPersistence:
if existing:
(err_id, ack, resolved_at, old_cat, old_severity, first_seen,
notif_sent, stored_suppression, acknowledged_at) = existing
notif_sent, stored_suppression, acknowledged_at, old_details_json) = existing
if ack == 1:
# SAFETY OVERRIDE: Critical CPU temperature ALWAYS re-triggers
@@ -680,6 +680,18 @@ class HealthPersistence:
conn.commit()
return event_info
# Original CPU policy is immutable for this row's incident.
# Never upgrade a legacy row from later/current settings.
if error_key == 'cpu_usage':
try:
old_details = json.loads(old_details_json or '{}')
except (ValueError, TypeError):
old_details = {}
details = dict(details) if isinstance(details, dict) else {}
details.pop('cpu_policy', None)
if isinstance(old_details, dict) and 'cpu_policy' in old_details:
details['cpu_policy'] = old_details['cpu_policy']
details_json = json.dumps(details)
# Not acknowledged - update existing active error
cursor.execute('''
UPDATE errors
@@ -776,7 +788,7 @@ class HealthPersistence:
# was created — otherwise "Storage 'Tuxis' unavailable"
# comes back as "Resolved - Storage" with no identity.
cursor.execute(
'SELECT details FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1',
'SELECT details, id, first_seen, resolved_at FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1',
(error_key,),
)
row = cursor.fetchone()
@@ -786,11 +798,18 @@ class HealthPersistence:
stored_details = json.loads(row[0])
except Exception:
stored_details = None
# Legacy/mismatched original policy cannot certify normality.
if error_key == 'cpu_usage' and (not isinstance(stored_details, dict)
or not isinstance(check_evidence, dict)
or check_evidence.get('policy') != stored_details.get('cpu_policy')
or not stored_details.get('cpu_policy')):
check_evidence = None
self._record_event(cursor, 'resolved', error_key, {
'reason': reason,
# Only explicit current-check callers attach this proof.
# Generic resolve/cleanup remains neutral.
'check_evidence': check_evidence,
'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': row[3]},
'entity': self._entity_from_details(stored_details),
'details': stored_details or {},
})
@@ -800,41 +819,40 @@ class HealthPersistence:
def get_recovery_evidence(self, error_key: str, first_seen: str):
"""Return fresh same-incident native check proof, never absence of errors.
Initially only the host CPU check has a stable condition identity. Other
Host CPU and exact per-service active checks carry provenance. Other
checks, generic clears, excluded/deleted records and legacy events stay
neutral until they have equivalent per-condition provenance.
"""
if error_key != 'cpu_usage' or not first_seen:
if not isinstance(error_key, str) or not first_seen or not (
error_key == 'cpu_usage' or error_key.startswith('pve_service_')):
return None
try:
with self._db_connection() as conn:
# One SQLite statement is one consistent row/ack/closure snapshot.
# Latest closure by event id, never search past a generic clear.
with self._db_lock, self._db_connection() as conn:
row = conn.execute('''
SELECT first_seen, last_seen, resolved_at, acknowledged
FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1
SELECT e.first_seen, e.last_seen, e.resolved_at, e.acknowledged,
e.id, v.timestamp, v.data
FROM errors e JOIN events v ON v.id = (
SELECT id FROM events WHERE error_key = e.error_key
AND event_type IN ('resolved', 'cleared') ORDER BY id DESC LIMIT 1
) WHERE e.error_key = ?
''', (error_key,)).fetchone()
if not row or row[0] != first_seen or not row[2] or row[3]:
return None
event = conn.execute('''
SELECT timestamp, data FROM events
WHERE error_key = ? AND event_type = 'resolved'
ORDER BY id DESC LIMIT 1
''', (error_key,)).fetchone()
if not event:
if not row or row[0] != first_seen or not row[2] or row[3]:
return None
proof = json.loads(event[1]).get('check_evidence')
if not isinstance(proof, dict) or proof.get('check') != error_key:
event_data = json.loads(row[6])
if event_data.get('incident') != {'id': row[4], 'first_seen': row[0], 'resolved_at': row[2]}:
return None
checked = proof.get('checked_at')
if not isinstance(checked, (int, float)) or isinstance(checked, bool):
proof = event_data.get('check_evidence')
from health_recovery import valid_check_evidence
if not valid_check_evidence(error_key, proof, now=datetime.now().timestamp()):
return None
checked = float(checked)
# Reuse the collector's existing two-hour freshness boundary.
now = datetime.now().timestamp()
if not 0 <= now - checked <= 7200:
if error_key == 'cpu_usage' and proof.get('policy') != event_data.get('details', {}).get('cpu_policy'):
return None
checked = float(proof['checked_at'])
last_seen = datetime.fromisoformat(row[1]).timestamp()
resolved = datetime.fromisoformat(row[2]).timestamp()
recorded = datetime.fromisoformat(event[0]).timestamp()
recorded = datetime.fromisoformat(row[5]).timestamp()
if not last_seen <= checked <= resolved <= recorded:
return None
return proof
@@ -906,7 +924,7 @@ class HealthPersistence:
return False
def clear_error(self, error_key: str):
def clear_error(self, error_key: str, *, check_evidence=None):
"""
Remove/resolve a specific error immediately.
Used when the condition that caused the error no longer exists
@@ -928,7 +946,7 @@ class HealthPersistence:
# Check if this error was acknowledged (dismissed)
cursor.execute('''
SELECT acknowledged FROM errors WHERE error_key = ?
SELECT acknowledged, id, first_seen FROM errors WHERE error_key = ?
''', (error_key,))
row = cursor.fetchone()
@@ -947,7 +965,10 @@ class HealthPersistence:
''', (now, error_key))
if cursor.rowcount > 0:
self._record_event(cursor, 'cleared', error_key, {'reason': 'condition_resolved'})
self._record_event(cursor, 'cleared', error_key, {
'reason': 'condition_resolved', 'check_evidence': check_evidence,
'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': now},
})
conn.commit()
+51
View File
@@ -0,0 +1,51 @@
"""Bounded recovery metadata admission, without importing monitor singletons.
Native persistence additionally binds the exact incident and closure. Manual
notifications remain authenticated caller assertions: shape validation cannot
establish that an asserted measurement actually happened.
"""
import math
import time
from typing import TypeGuard
def _finite_number(value) -> TypeGuard[int | float]:
try:
return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value)
except (OverflowError, ValueError, TypeError):
return False
def valid_check_evidence(error_key, proof, *, now=None):
"""Validate a supported measurement contract, not its external authenticity."""
if not isinstance(proof, dict) or proof.get('check') != error_key:
return False
checked = proof.get('checked_at')
now = time.time() if now is None else now
if not _finite_number(checked) or not _finite_number(now) or not 0 <= now-checked <= 7200:
return False
if error_key == 'cpu_usage':
policy = proof.get('policy')
if not isinstance(policy, dict) or set(policy) != {'warning', 'critical', 'recovery'}:
return False
if not all(_finite_number(v) and 1 <= v <= 100 for v in policy.values()):
return False
if policy['warning'] > policy['critical']:
return False
value, maximum, count = proof.get('value'), proof.get('max_sample'), proof.get('normal_samples')
return (_finite_number(value) and _finite_number(maximum)
and 0 <= value <= maximum < min(policy['warning'], policy['recovery'])
and isinstance(count, int) and not isinstance(count, bool) and count >= 10)
if isinstance(error_key, str) and error_key.startswith('pve_service_'):
service = error_key[len('pve_service_'):]
return (bool(service) and proof.get('service') == service
and proof.get('state') == 'active'
and type(proof.get('returncode')) is int and proof['returncode'] == 0)
return False
def presents_recovery(data):
"""One presentation predicate shared by template, icon and email badge."""
return (isinstance(data, dict) and data.get('recovery_outcome') == 'resolved'
and data.get('is_recovery') is True
and valid_check_evidence(data.get('error_key'), data.get('check_evidence')))
+7 -6
View File
@@ -1038,10 +1038,8 @@ class EmailChannel(NotificationChannel):
# Determine group for section header
event_type = data.get('_event_type', '')
if event_type == 'error_resolved':
if (data.get('recovery_outcome') == 'resolved'
and data.get('is_recovery') is True
and isinstance(data.get('check_evidence'), dict)
and data['check_evidence'].get('check') == 'cpu_usage'):
from health_recovery import presents_recovery
if presents_recovery(data):
sev.update(self._SEV_STYLE['OK'])
sev['label'] = _runtime_notification_text('healthRecovery.status', data)
else:
@@ -1070,8 +1068,11 @@ class EmailChannel(NotificationChannel):
or backup_email or data.get('_restore_summary') or data.get('_backup_summary'))
temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else ''
temp_table_layout = 'table-layout:fixed;' if wrap_body else ''
backup_title_wrap = temp_cell_wrap if backup_email else ''
backup_metadata_layout = 'table-layout:fixed;' if backup_email else ''
# Recovery exposes the same literal host context as backup notices.
# Keep wrapping event-scoped; unrelated mail remains byte-identical.
context_email = backup_email or event_type == 'error_resolved'
backup_title_wrap = temp_cell_wrap if context_email else ''
backup_metadata_layout = 'table-layout:fixed;' if context_email else ''
section_label = _runtime_text(f'email.groups.{group}', data)
report_label = _runtime_text('email.report', data, group=section_label)
host_label = _runtime_text('email.host', data)
+17 -18
View File
@@ -2070,10 +2070,8 @@ def render_template(event_type: str, data: Dict[str, Any],
return ''
safe_vars = _SafeDict(variables)
if (event_type == 'error_resolved' and data.get('recovery_outcome') == 'resolved'
and data.get('is_recovery') is True
and isinstance(data.get('check_evidence'), dict)
and data['check_evidence'].get('check') == 'cpu_usage'):
from health_recovery import presents_recovery
if event_type == 'error_resolved' and presents_recovery(data):
safe_vars['_health_title'] = runtime_message('healthRecovery.title', language, **variables)
safe_vars['_health_body'] = runtime_message('healthRecovery.body', language, **variables)
template['title'] = '{_health_title}'
@@ -2089,13 +2087,14 @@ def render_template(event_type: str, data: Dict[str, Any],
# parse the table/logs and format a rich body instead of the sparse template.
pve_message = data.get('pve_message', '')
backup_diagnostics = []
principal_cause = None
def bounded_backup_diagnostics(lines):
def bounded_backup_diagnostics(lines, principal_cause=None):
# 1024 chars matches the repository's small-channel message convention;
# 8 lines keeps repeated producer warnings readable. Inventory/title
# size is separate: this is not a one-Telegram-message guarantee.
unique = list(dict.fromkeys(line for line in lines if line.strip()))
principal = next((line for line in unique if re.search(r'\b(?:ERROR:|TASK ERROR:)', line, re.IGNORECASE)), None)
principal = principal_cause or next((line for line in unique if re.search(r'\b(?:ERROR:|TASK ERROR:)', line, re.IGNORECASE)), None)
if principal:
unique.remove(principal)
unique.insert(0, principal)
@@ -2189,16 +2188,18 @@ def render_template(event_type: str, data: Dict[str, Any],
source_subject = cause.group(1).strip() if cause else ''
if source_subject.lower() == 'multiple problems':
source_subject = ''
cause_in_diagnostics = any(
line.strip() == source_subject or
re.split(r'\b(?:TASK ERROR:|ERROR:)\s*', line, maxsplit=1, flags=re.IGNORECASE)[-1].strip() == source_subject
for line in backup_diagnostics)
if source_subject and source_subject not in body_text and not cause_in_diagnostics:
# A unique job/setup cause must survive a warning-heavy report.
backup_diagnostics.insert(0, source_subject)
if source_subject and source_subject not in body_text:
# Reserve the subject-equivalent diagnostic BEFORE the cap. Finding
# it in uncapped logs is not enough: that late line could be omitted.
principal_cause = next((line for line in backup_diagnostics
if line.strip() == source_subject or
re.split(r'\b(?:TASK ERROR:|ERROR:)\s*', line, maxsplit=1, flags=re.IGNORECASE)[-1].strip() == source_subject), None)
if not principal_cause:
principal_cause = source_subject
backup_diagnostics.insert(0, source_subject)
if backup_diagnostics:
body_text += '\n' + bounded_backup_diagnostics(backup_diagnostics)
body_text += '\n' + bounded_backup_diagnostics(backup_diagnostics, principal_cause)
# Clean up: collapse runs of 3+ blank lines into 1, remove trailing whitespace
import re as _re
@@ -2531,10 +2532,8 @@ def enrich_with_emojis(event_type: str, title: str, body: str,
severity = data.get('severity', 'INFO')
icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '')
if (event_type == 'error_resolved' and data.get('recovery_outcome') == 'resolved'
and data.get('is_recovery') is True
and isinstance(data.get('check_evidence'), dict)
and data['check_evidence'].get('check') == 'cpu_usage'):
from health_recovery import presents_recovery
if event_type == 'error_resolved' and presents_recovery(data):
icon = '✅'
if event_type == 'backup_complete':
icon = {