mirror of
https://github.com/MacRimi/ProxMenux.git
synced 2026-10-09 06:56:37 +00:00
fix(notifications): scope outcome improvements to backups
This commit is contained in:
@@ -1,86 +0,0 @@
|
|||||||
"""Pinned, inert recovery review. No operational module imports or threads.
|
|
||||||
Run with python3 -S under bwrap --unshare-net; DBs and export use TMPDIR.
|
|
||||||
"""
|
|
||||||
import ast, contextlib, datetime, hashlib, io, json, os, pathlib, sqlite3, subprocess, sys, tarfile, tempfile, threading, types, typing
|
|
||||||
from unittest.mock import patch
|
|
||||||
REPO=pathlib.Path('/home/martino/projects/proxmox/notification-maintainer-followup')
|
|
||||||
BASE=datetime.datetime(2026,9,30,20,0,0).timestamp()
|
|
||||||
class Clock(datetime.datetime):
|
|
||||||
epoch=BASE
|
|
||||||
tick=0.0
|
|
||||||
@classmethod
|
|
||||||
def now(cls,tz=None):
|
|
||||||
value=cls.fromtimestamp(cls.epoch,tz)
|
|
||||||
cls.epoch+=cls.tick
|
|
||||||
return value
|
|
||||||
TIME=types.SimpleNamespace(time=lambda:Clock.epoch)
|
|
||||||
|
|
||||||
def extract(path,name,owner,ns):
|
|
||||||
tree=ast.parse(path.read_text())
|
|
||||||
nodes=tree.body if owner is None else next(n.body for n in tree.body if isinstance(n,ast.ClassDef) and n.name==owner)
|
|
||||||
node=next(n for n in nodes if isinstance(n,(ast.FunctionDef,ast.AsyncFunctionDef)) and n.name==name)
|
|
||||||
node.decorator_list=[]
|
|
||||||
exec(compile(ast.Module(body=[node],type_ignores=[]),str(path),'exec'),ns)
|
|
||||||
return ns[name]
|
|
||||||
|
|
||||||
from notification_fixture import SCRIPTS as scripts
|
|
||||||
pns=dict(vars(typing),datetime=Clock,timedelta=datetime.timedelta,json=json,sqlite3=sqlite3,contextmanager=contextlib.contextmanager,re=__import__('re'),_re_disk_base=__import__('re'))
|
|
||||||
pns['disk_base_name']=extract(scripts/'health_persistence.py','disk_base_name',None,pns)
|
|
||||||
methods={n:extract(scripts/'health_persistence.py',n,'HealthPersistence',pns) for n in ('_get_conn','_db_connection','_init_database','record_error','_record_error_impl','resolve_error','_resolve_error_impl','get_recovery_evidence','_record_event','_entity_from_details','clear_error','get_active_errors','is_error_active','is_error_acknowledged','_get_setting_impl','get_setting','set_setting','acknowledge_error','_acknowledge_error_impl','get_excluded_interface_names')}
|
|
||||||
def make_store(directory):
|
|
||||||
s=types.SimpleNamespace(db_path=pathlib.Path(directory)/'health.sqlite',_db_lock=threading.RLock(),DEFAULT_SUPPRESSION_HOURS=24,CATEGORY_SETTING_MAP={})
|
|
||||||
for n,f in methods.items():
|
|
||||||
if n=='_entity_from_details':setattr(s,n,f)
|
|
||||||
elif n=='_db_connection':setattr(s,n,types.MethodType(contextlib.contextmanager(f),s))
|
|
||||||
else:setattr(s,n,types.MethodType(f,s))
|
|
||||||
s._init_database()
|
|
||||||
return s
|
|
||||||
def sql(s,query,args=()):
|
|
||||||
with s._db_connection() as c:
|
|
||||||
data=c.execute(query,args).fetchall();c.commit();return data
|
|
||||||
def record(s,key='cpu_usage',category='cpu',reason='CPU high',details=None):
|
|
||||||
Clock.epoch=BASE-600
|
|
||||||
with patch.dict(sys.modules,{'os':types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:True))}):s.record_error(key,category,'WARNING',reason,details)
|
|
||||||
Clock.epoch=BASE-10
|
|
||||||
with patch.dict(sys.modules,{'os':types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:True))}):s.record_error(key,category,'WARNING',reason,details)
|
|
||||||
Clock.epoch=BASE
|
|
||||||
assert s.get_active_errors()
|
|
||||||
return sql(s,'SELECT first_seen FROM errors WHERE error_key=?',(key,))[0][0]
|
|
||||||
def cpu(s,current=20,history=None,warning=85,critical=95):
|
|
||||||
ns=dict(vars(typing),time=TIME,os=types.SimpleNamespace(cpu_count=lambda:4),health_persistence=s,psutil=types.SimpleNamespace(cpu_percent=lambda **kw:current,cpu_count=lambda:4))
|
|
||||||
fn=extract(scripts/'health_monitor.py','_check_cpu_with_hysteresis','HealthMonitor',ns)
|
|
||||||
if history is None:history=[{'value':20,'time':BASE-i*10} for i in range(1,11)]
|
|
||||||
target=types.SimpleNamespace(state_history={'cpu_usage':list(history)},CPU_WARNING=85,CPU_CRITICAL=95,CPU_RECOVERY=75,CPU_WARNING_DURATION=300,CPU_CRITICAL_DURATION=300,CPU_RECOVERY_DURATION=120,_check_cpu_temperature=lambda:None)
|
|
||||||
refresh=extract(scripts/'health_monitor.py','_refresh_thresholds','HealthMonitor',{})
|
|
||||||
with patch.dict(sys.modules,{'health_thresholds':types.SimpleNamespace(get=lambda section,key:({'warning':warning,'critical':critical}.get(key) if section=='cpu' else None))}):refresh(target)
|
|
||||||
return fn(target)
|
|
||||||
def poll(s,first,key='cpu_usage',category='cpu',reason='CPU high',details=None,first_done=True,foreign=False,restored=False):
|
|
||||||
events=[]
|
|
||||||
meta={'category':category,'reason':reason,'severity':'WARNING','first_seen':first,'details':details}
|
|
||||||
c=types.SimpleNamespace(_hostname='node-a',_ENTITY_MAP={'cpu':('node',''),'pve_services':('node',''),'network':('node','')},_first_poll_done=first_done,_known_errors={key:meta},_notified_severity={key:'WARNING'},_last_notified={key:BASE-1},SAME_ERROR_COOLDOWN=86400,_get_cooldown_from_db=lambda *a:BASE-1,_queue=types.SimpleNamespace(put=events.append),_guest_storage_error_is_now_foreign=lambda *a:foreign,_save_known_errors_meta=lambda:None)
|
|
||||||
ns=dict(vars(typing),time=TIME,json=json,re=__import__('re'),NotificationEvent=lambda *a,**kw:types.SimpleNamespace(event_type=a[0],severity=a[1],data=a[2],**kw),startup_grace=types.SimpleNamespace(should_suppress_category=lambda *a:False))
|
|
||||||
c._guest_storage_error_is_now_foreign=extract(scripts/'notification_events.py','_guest_storage_error_is_now_foreign','PollingCollector',ns)
|
|
||||||
fn=extract(scripts/'notification_events.py','_check_persistent_health','PollingCollector',ns)
|
|
||||||
with patch.dict(sys.modules,{'health_persistence':types.SimpleNamespace(health_persistence=s),'flask_server':types.SimpleNamespace(get_proxmox_node_name=lambda:'node-a',get_cached_pvesh_cluster_resources_vm=lambda:[{'vmid':100,'type':'lxc','node':'node-b' if foreign else 'node-a'}]),'datetime':types.SimpleNamespace(**{**vars(datetime),'datetime':Clock})}):
|
|
||||||
if restored:
|
|
||||||
c._KNOWN_ERRORS_SETTING_KEY='pollingcollector_known_errors_v1'
|
|
||||||
s.set_setting(c._KNOWN_ERRORS_SETTING_KEY,json.dumps(c._known_errors));c._known_errors={}
|
|
||||||
extract(scripts/'notification_events.py','_load_known_errors_meta','PollingCollector',ns)(c)
|
|
||||||
c._first_poll_done=bool(c._known_errors)
|
|
||||||
fn(c)
|
|
||||||
return [e.data for e in events],c
|
|
||||||
def service(store, rc=0, stdout='active\n', raised=False, services=('pvedaemon',), clustered=False):
|
|
||||||
calls=[]
|
|
||||||
def run(argv, **kw):
|
|
||||||
calls.append((argv, kw))
|
|
||||||
assert argv[:2] == ['systemctl', 'is-active']
|
|
||||||
if raised: raise TimeoutError('inert timeout')
|
|
||||||
return types.SimpleNamespace(returncode=rc, stdout=stdout)
|
|
||||||
ns=dict(vars(typing),time=TIME,os=types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:clustered)),subprocess=types.SimpleNamespace(run=run),health_persistence=store)
|
|
||||||
result=extract(scripts/'health_monitor.py','_check_pve_services','HealthMonitor',ns)(types.SimpleNamespace(PVE_SERVICES=list(services)))
|
|
||||||
return result,calls
|
|
||||||
|
|
||||||
@contextlib.contextmanager
|
|
||||||
def case():
|
|
||||||
Clock.epoch=BASE
|
|
||||||
with tempfile.TemporaryDirectory(prefix='recovery-review-db-',dir=os.environ.get('TMPDIR')) as d:yield make_store(d)
|
|
||||||
@@ -138,14 +138,10 @@ class CommandDescriptionsTests(unittest.TestCase):
|
|||||||
source = catalog('en')['runtime']['notifications']
|
source = catalog('en')['runtime']['notifications']
|
||||||
for key, value in source['backup'].items():
|
for key, value in source['backup'].items():
|
||||||
local.setdefault('backup', {}).setdefault(key, value)
|
local.setdefault('backup', {}).setdefault(key, value)
|
||||||
local['channels']['email']['severity'].setdefault(
|
|
||||||
'observation', source['channels']['email']['severity']['observation'])
|
|
||||||
local['channels']['email']['status'].setdefault(
|
local['channels']['email']['status'].setdefault(
|
||||||
'unconfirmed', source['channels']['email']['status']['unconfirmed'])
|
'unconfirmed', source['channels']['email']['status']['unconfirmed'])
|
||||||
local['channels']['email']['status'].setdefault(
|
local['channels']['email']['status'].setdefault(
|
||||||
'completed_with_warnings', source['channels']['email']['status']['completed_with_warnings'])
|
'completed_with_warnings', source['channels']['email']['status']['completed_with_warnings'])
|
||||||
for key, value in source['healthRecovery'].items():
|
|
||||||
local.setdefault('healthRecovery', {}).setdefault(key, value)
|
|
||||||
path.write_text(json.dumps(temporary, ensure_ascii=False))
|
path.write_text(json.dumps(temporary, ensure_ascii=False))
|
||||||
# Model steady state after the bot fills these intentional new
|
# Model steady state after the bot fills these intentional new
|
||||||
# messages; keep repository locales and all other leaves intact.
|
# messages; keep repository locales and all other leaves intact.
|
||||||
|
|||||||
@@ -0,0 +1,116 @@
|
|||||||
|
"""Backup-only maintainer contract at actual render/dispatch/email seams.
|
||||||
|
|
||||||
|
Scope-approved seams: receiver, template lookup, rich enrichment, inert queued
|
||||||
|
and manual delivery with an email capture sink. No host operations or sends.
|
||||||
|
"""
|
||||||
|
import unittest
|
||||||
|
from notification_fixture import templates, LANGUAGES
|
||||||
|
from notification_final_fixture import deliver
|
||||||
|
|
||||||
|
|
||||||
|
class BackupSplitTests(unittest.TestCase):
|
||||||
|
def test_recovery_default_is_upstream_resolved_without_proof(self):
|
||||||
|
for language in LANGUAGES:
|
||||||
|
data = {'hostname': 'alias', 'category': 'temperature',
|
||||||
|
'reason': 'Temperature high (recovered)', 'duration': '2m',
|
||||||
|
'original_severity': 'WARNING'}
|
||||||
|
rendered = templates.render_template('error_resolved', data, language)
|
||||||
|
expected = templates.runtime_message('templates.error_resolved.title', language,
|
||||||
|
hostname='alias', category='temperature', entity_suffix='')
|
||||||
|
self.assertEqual(rendered['title'], expected)
|
||||||
|
rich, _ = templates.enrich_with_emojis('error_resolved', rendered['title'], rendered['body'], data)
|
||||||
|
self.assertTrue(rich.startswith('✅ '))
|
||||||
|
result = deliver('error_resolved', data, 'OK', language)
|
||||||
|
self.assertIn('background:#f0fdf4;', result['html'])
|
||||||
|
|
||||||
|
def test_pre_guest_native_subject_keeps_only_cause_once_with_display_alias(self):
|
||||||
|
from notification_fixture import receive
|
||||||
|
raw_host = 'pve-production.internal.example'
|
||||||
|
message = ('Details\n=======\nVMID Name Status Time Size Filename\n'
|
||||||
|
'\nTotal running time: 0s\nTotal size: 0 B')
|
||||||
|
for raw in (message, 'ERROR: unable to activate storage PBS\n' + message):
|
||||||
|
event = receive(raw, 'error', 'vzdump backup status (' + raw_host +
|
||||||
|
'): backup failed: unable to activate storage PBS')
|
||||||
|
event.data['hostname'] = 'display-alias {rack.location}'
|
||||||
|
for language in LANGUAGES:
|
||||||
|
for manual in (False, True):
|
||||||
|
result = deliver(event.event_type, event.data, event.severity, language, manual=manual)
|
||||||
|
self.assertNotIn(raw_host, result['text'])
|
||||||
|
self.assertNotIn('vzdump backup status', result['text'])
|
||||||
|
self.assertEqual(result['text'].count('unable to activate storage PBS'), 1)
|
||||||
|
self.assertIn('display-alias {rack.location}', result['title'])
|
||||||
|
self.assertEqual(result['data']['pve_title'], event.data['pve_title'])
|
||||||
|
self.assertEqual(result['data']['pve_message'], raw)
|
||||||
|
# No global hostname or guest-name substitution: actual diagnostics are
|
||||||
|
# authoritative, even when they happen to contain the original hostname.
|
||||||
|
event = receive(message, 'error', 'vzdump backup status (' + raw_host +
|
||||||
|
'): backup failed: cannot connect to ' + raw_host)
|
||||||
|
result = deliver(event.event_type, {**event.data, 'hostname': 'alias'}, event.severity)
|
||||||
|
self.assertEqual(result['body'].count('cannot connect to ' + raw_host), 1)
|
||||||
|
self.assertNotIn('vzdump backup status', result['body'])
|
||||||
|
|
||||||
|
def test_backup_legacy_keys_keep_meaning_and_unknown_uses_new_keys(self):
|
||||||
|
# Frozen upstream contract; no Git history required in shipped tests.
|
||||||
|
legacy = {'en': {'title': '{hostname} → {storage}: Backup complete — {vmname} ({vmid})', 'body': 'Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}', 'label': 'Backup complete'}, 'de': {'title': '{hostname} → {storage}: Sicherung abgeschlossen – {vmname} ({vmid})', 'body': 'Die Sicherung von {vmname} (ID: {vmid}) wurde am {storage} erfolgreich abgeschlossen.\nGröße: {size}', 'label': 'Sicherung abgeschlossen'}, 'es': {'title': '{hostname} → {storage}: Backup completado — {vmname} ({vmid})', 'body': 'El backup de {vmname} (ID: {vmid}) se ha completado correctamente en {storage}.\nTamaño: {size}', 'label': 'Backup completado'}, 'fr': {'title': '{hostname} → {storage}\xa0: Sauvegarde terminée — {vmname} ({vmid})', 'body': "La sauvegarde de {vmname} (ID\xa0: {vmid}) s'est terminée avec succès le {storage}.\nTaille\xa0: {size}", 'label': 'Sauvegarde terminée'}, 'it': {'title': '{hostname} → {storage}: Backup completato — {vmname} ({vmid})', 'body': 'Backup di {vmname} (ID: {vmid}) completato con successo su {storage}.\nTaglia: {size}', 'label': 'Backup completato'}, 'pt': {'title': '{hostname} → {storage}: Backup concluído — {vmname} ({vmid})', 'body': 'Backup de {vmname} (ID: {vmid}) concluído com sucesso em {storage}.\nTamanho: {size}', 'label': 'Backup concluído'}, 'sk': {'title': '{hostname} → {storage}: Záloha dokončená — {vmname} ({vmid})', 'body': 'Záloha {vmname} (ID: {vmid}) na úložisku {storage} bola úspešne dokončená.\nVeľkosť: {size}', 'label': 'Záloha bola dokončená'}, 'sv': {'title': '{hostname} → {storage}: Säkerhetskopiering klar — {vmname} ({vmid})', 'body': 'Säkerhetskopiering av {vmname} (ID: {vmid}) slutfördes framgångsrikt på {storage}.\nStorlek: {size}', 'label': 'Säkerhetskopieringen är klar'}}
|
||||||
|
for language, expected in legacy.items():
|
||||||
|
self.assertEqual(templates._load_runtime_catalog(language)['templates']['backup_complete'], expected)
|
||||||
|
result = templates.render_template('backup_complete', {'hostname': 'alias {rack.location}'}, language)
|
||||||
|
self.assertIn('alias {rack.location}', result['title'])
|
||||||
|
self.assertNotIn('()', result['title'])
|
||||||
|
new_title = templates.runtime_message('backup.unconfirmedTitle', language, hostname='alias {rack.location}')
|
||||||
|
self.assertTrue(new_title)
|
||||||
|
self.assertEqual(result['title'], new_title)
|
||||||
|
self.assertIn(templates.runtime_message('backup.unconfirmedBody', language), result['body'])
|
||||||
|
# Absent/blank/non-string translation uses English per-key fallback.
|
||||||
|
from unittest.mock import patch
|
||||||
|
english = templates._load_runtime_catalog('en')
|
||||||
|
for value in (None, '', {}, []):
|
||||||
|
missing = {'backup': {'unconfirmedTitle': value, 'unconfirmedBody': value}}
|
||||||
|
with patch.object(templates, '_load_runtime_catalog', side_effect=lambda lang: english if lang == 'en' else missing):
|
||||||
|
result = templates.render_template('backup_complete', {'hostname': 'alias'}, 'it')
|
||||||
|
self.assertEqual(result['title'], 'alias: Backup outcome unconfirmed')
|
||||||
|
self.assertEqual(result['body'], 'The backup outcome is not confirmed.')
|
||||||
|
# Both catalogs missing: new outcome text still has explicit EN defaults.
|
||||||
|
with patch.object(templates, '_load_runtime_catalog', return_value={}):
|
||||||
|
result = templates.render_template('backup_complete', {'hostname': 'alias'}, 'it')
|
||||||
|
self.assertEqual(result['title'], 'alias: Backup outcome unconfirmed')
|
||||||
|
self.assertEqual(result['body'], 'The backup outcome is not confirmed.')
|
||||||
|
|
||||||
|
def test_restore_keeps_original_ready_line_and_success_icon_with_warnings(self):
|
||||||
|
from notification_final_fixture import restore_event
|
||||||
|
ready = {'en': 'The node is now fully ready to use.', 'de': 'Der Knoten ist nun vollständig einsatzbereit.', 'es': 'El nodo está listo para usarse.', 'fr': 'Le nœud est maintenant entièrement prêt à être utilisé.', 'it': "Il nodo è ora completamente pronto per l'uso.", 'pt': 'O nó agora está totalmente pronto para uso.', 'sk': 'Uzol je teraz úplne pripravený na použitie.', 'sv': 'Noden är nu helt redo att användas.'}
|
||||||
|
for warning in ('', 'missing module zfs'):
|
||||||
|
event = restore_event(warning)
|
||||||
|
for language in LANGUAGES:
|
||||||
|
result = deliver(event['event_type'], event['data'], event['severity'], language)
|
||||||
|
self.assertTrue(result['title'].startswith('✅ '))
|
||||||
|
self.assertIn(ready[language], result['body'])
|
||||||
|
self.assertIn(ready[language], result['text'])
|
||||||
|
self.assertIn('2m', result['text'])
|
||||||
|
if warning:
|
||||||
|
self.assertIn(warning, result['text'])
|
||||||
|
quiet = deliver(event['event_type'], event['data'], event['severity'], language, quiet=True)
|
||||||
|
self.assertIn('✅', quiet['body'])
|
||||||
|
self.assertIn(' ' + ready[language], quiet['body'])
|
||||||
|
self.assertIn('white-space:pre-wrap;', quiet['html'])
|
||||||
|
|
||||||
|
def test_spanish_outcome_and_restore_titles_are_capitalized_and_failure_is_exact(self):
|
||||||
|
expected = {'confirmed': 'Backup completado', 'completed_with_warnings': 'Backup completado con advertencias',
|
||||||
|
'unconfirmed': 'Resultado del backup sin confirmar', 'failed': 'Backup fallido'}
|
||||||
|
for outcome, title in expected.items():
|
||||||
|
result = templates.render_template('backup_complete', {'hostname': 'alias', 'backup_outcome': outcome}, 'es')
|
||||||
|
self.assertEqual(result['title'], 'alias: ' + title)
|
||||||
|
failure = templates.render_template('backup_fail', {'hostname': 'alias'}, 'es')
|
||||||
|
self.assertEqual(failure['title'], 'alias: Backup fallido')
|
||||||
|
restore = templates.render_template('system_restore_completed', {'hostname': 'alias'}, 'es')
|
||||||
|
self.assertEqual(restore['title'], 'alias: Restauración del host finalizada')
|
||||||
|
|
||||||
|
def test_recovery_only_quiet_release_keeps_upstream_summary_contract(self):
|
||||||
|
result = deliver('error_resolved', {'hostname': 'alias', 'category': 'temperature',
|
||||||
|
'reason': 'old (recovered)', 'duration': '2m'}, 'OK', quiet=True)
|
||||||
|
self.assertIn(templates.runtime_message('digest.lead', 'en', count=1).strip(), result['body'])
|
||||||
|
self.assertIn(templates.runtime_message('digest.footer', 'en'), result['body'])
|
||||||
|
self.assertIn('✅', result['body'])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__': unittest.main()
|
||||||
@@ -54,8 +54,6 @@ class CorrectionTests(unittest.TestCase):
|
|||||||
probe.catalogs = {lang: copy.deepcopy(templates._load_runtime_catalog(lang)) for lang in LANGUAGES}
|
probe.catalogs = {lang: copy.deepcopy(templates._load_runtime_catalog(lang)) for lang in LANGUAGES}
|
||||||
parity(probe) # shipped missing keys remain allowed
|
parity(probe) # shipped missing keys remain allowed
|
||||||
probe.catalogs['sk']['backup'] = copy.deepcopy(probe.catalogs['en']['backup'])
|
probe.catalogs['sk']['backup'] = copy.deepcopy(probe.catalogs['en']['backup'])
|
||||||
for key in ('observation',):
|
|
||||||
probe.catalogs['sk']['channels']['email']['severity'][key] = probe.catalogs['en']['channels']['email']['severity'][key]
|
|
||||||
probe.catalogs['sk']['channels']['email']['status']['unconfirmed'] = probe.catalogs['en']['channels']['email']['status']['unconfirmed']
|
probe.catalogs['sk']['channels']['email']['status']['unconfirmed'] = probe.catalogs['en']['channels']['email']['status']['unconfirmed']
|
||||||
parity(probe) # generation of exactly the pending keys is legal
|
parity(probe) # generation of exactly the pending keys is legal
|
||||||
probe.catalogs['sk']['backup']['confirmedTitle'] = 'Missing hostname token'
|
probe.catalogs['sk']['backup']['confirmedTitle'] = 'Missing hostname token'
|
||||||
@@ -63,59 +61,14 @@ class CorrectionTests(unittest.TestCase):
|
|||||||
|
|
||||||
|
|
||||||
def test_actual_neutral_style_is_not_success_green(self):
|
def test_actual_neutral_style_is_not_success_green(self):
|
||||||
for event, severity, data in (('error_resolved', 'OK', {}),
|
for event, severity, data in (('backup_complete', 'INFO', {'backup_outcome': 'unconfirmed'}),):
|
||||||
('backup_complete', 'INFO', {'backup_outcome': 'unconfirmed'})):
|
|
||||||
result, markup = email(event, data, severity)
|
result, markup = email(event, data, severity)
|
||||||
self.assertIn('background:#f9fafb;', markup)
|
self.assertIn('background:#f9fafb;', markup)
|
||||||
self.assertNotIn('background:#f0fdf4;', markup)
|
self.assertNotIn('background:#f0fdf4;', markup)
|
||||||
|
|
||||||
|
|
||||||
def test_actual_manual_caller_carries_event_presentation_context(self):
|
|
||||||
from notification_fixture import extract, SCRIPTS, EmailChannel
|
|
||||||
from typing import Optional, Dict, Any
|
|
||||||
from threading import Lock
|
|
||||||
captured = []
|
|
||||||
channel = object.__new__(EmailChannel)
|
|
||||||
channel.subject_prefix = '[ProxMenux]'
|
|
||||||
class Sink:
|
|
||||||
def send(self, title, body, severity, data):
|
|
||||||
captured.append((data, channel._format_html(title, body, severity, data)))
|
|
||||||
return {'success': True}
|
|
||||||
ns = {'Optional': Optional, 'Dict': Dict, 'Any': Any, 'TEMPLATES': templates.TEMPLATES,
|
|
||||||
'resolve_notification_hostname': lambda host, config: host or 'node-a',
|
|
||||||
'render_template': templates.render_template, '_should_bypass_ai': lambda event: True}
|
|
||||||
send = extract(SCRIPTS / 'notification_manager.py', 'send_notification', 'NotificationManager', ns)
|
|
||||||
class Manager:
|
|
||||||
_channels = {'email': Sink()}
|
|
||||||
_config = {}
|
|
||||||
_lock = Lock()
|
|
||||||
def _notification_language(self): return 'en'
|
|
||||||
def is_event_enabled(self, event): return True
|
|
||||||
def _build_ai_config(self): return {}
|
|
||||||
def _record_history(self, *args): pass
|
|
||||||
data = {'category': 'temperature', 'reason': 'old', 'duration': '3d', 'original_severity': 'WARNING',
|
|
||||||
'_event_type': 'node_reconnect', '_group': 'cluster'}
|
|
||||||
result = send(Manager(), 'error_resolved', 'OK', '', '', data)
|
|
||||||
self.assertTrue(result['success'])
|
|
||||||
context, markup = captured[0]
|
|
||||||
self.assertEqual(context['_event_type'], 'error_resolved')
|
|
||||||
self.assertEqual(context['_group'], 'health')
|
|
||||||
self.assertIn('NO LONGER REPORTED', markup)
|
|
||||||
self.assertNotIn('>RESOLVED</span>', markup)
|
|
||||||
self.assertNotIn('color:#16a34a', markup)
|
|
||||||
self.assertEqual(data['_event_type'], 'node_reconnect') # caller not mutated
|
|
||||||
|
|
||||||
|
|
||||||
def test_disappearance_body_keeps_observation_age_not_green_severity(self):
|
|
||||||
data = {'hostname': 'node-a', 'category': 'temperature', 'reason': 'old observation',
|
|
||||||
'duration': '3d 2h', 'original_severity': 'WARNING', 'severity': 'OK'}
|
|
||||||
for language in LANGUAGES:
|
|
||||||
result, markup = email('error_resolved', data, 'OK', language)
|
|
||||||
for line in result['body'].splitlines():
|
|
||||||
if line.strip(): self.assertIn(line.strip(), html.unescape(markup))
|
|
||||||
self.assertNotIn('>OK</span>', markup)
|
|
||||||
self.assertNotIn('color:#16a34a', markup)
|
|
||||||
self.assertNotIn('>RESOLVED</span>', markup)
|
|
||||||
|
|
||||||
|
|
||||||
def test_actual_restore_endpoint_warnings_and_counts_reach_email(self):
|
def test_actual_restore_endpoint_warnings_and_counts_reach_email(self):
|
||||||
|
|||||||
@@ -77,23 +77,13 @@ class FinalCorrectionsTests(unittest.TestCase):
|
|||||||
self.assertNotIn('script', result['tags'])
|
self.assertNotIn('script', result['tags'])
|
||||||
|
|
||||||
|
|
||||||
def test_long_disappearance_reason_is_present_once_in_actual_dispatch(self):
|
|
||||||
reason = 'Temperature exceeded configured limit; the source stopped reporting this observation after expiry.'
|
|
||||||
for language in LANGUAGES:
|
|
||||||
for manual in (False, True):
|
|
||||||
result = deliver('error_resolved', {'hostname':'node-a','category':'temperature',
|
|
||||||
'reason':reason,'duration':'3d 2h','original_severity':'WARNING'}, 'OK', language, manual=manual)
|
|
||||||
self.assertEqual(result['text'].count(reason), 1)
|
|
||||||
self.assertNotIn('>OK</span>', result['html'])
|
|
||||||
self.assertNotIn('>RESOLVED</span>', result['html'])
|
|
||||||
|
|
||||||
|
|
||||||
def test_raw_restore_and_observation_cells_use_event_scoped_mail_wrapping(self):
|
def test_raw_restore_cells_use_event_scoped_mail_wrapping(self):
|
||||||
token = 'b' * 64
|
token = 'b' * 64
|
||||||
event = restore_event('Boot check: recorded token ' + token + '; verification pending')
|
event = restore_event('Boot check: recorded token ' + token + '; verification pending')
|
||||||
for language in LANGUAGES:
|
for language in LANGUAGES:
|
||||||
results = [deliver(event['event_type'],event['data'],event['severity'],language,quiet=quiet) for quiet in (False,True)]
|
results = [deliver(event['event_type'],event['data'],event['severity'],language,quiet=quiet) for quiet in (False,True)]
|
||||||
results.append(deliver('error_resolved',{'hostname':'node-a','reason':token,'category':'temperature','duration':'3d 2h'},'OK',language))
|
|
||||||
for result in results:
|
for result in results:
|
||||||
self.assertIn('table-layout:fixed;', result['html'])
|
self.assertIn('table-layout:fixed;', result['html'])
|
||||||
self.assertIn('word-wrap:break-word;', result['html'])
|
self.assertIn('word-wrap:break-word;', result['html'])
|
||||||
|
|||||||
@@ -61,7 +61,7 @@ class MaintainerFollowupTests(unittest.TestCase):
|
|||||||
report+'\nINFO: Starting Backup of VM 100 (lxc)'):
|
report+'\nINFO: Starting Backup of VM 100 (lxc)'):
|
||||||
self.assertEqual(templates._parse_vzdump_message(message)['vms'][0]['type'],'')
|
self.assertEqual(templates._parse_vzdump_message(message)['vms'][0]['type'],'')
|
||||||
|
|
||||||
def test_original_subject_only_retained_when_no_guest_context(self):
|
def test_native_subject_keeps_cause_without_host_envelope(self):
|
||||||
subject = 'vzdump backup status (raw-host): backup failed: multiple problems'
|
subject = 'vzdump backup status (raw-host): backup failed: multiple problems'
|
||||||
event = receive('ERROR: archive write failed\n'+NATIVE_REPORT,'error',subject)
|
event = receive('ERROR: archive write failed\n'+NATIVE_REPORT,'error',subject)
|
||||||
event.data['hostname']='configured-alias'
|
event.data['hostname']='configured-alias'
|
||||||
@@ -72,7 +72,8 @@ class MaintainerFollowupTests(unittest.TestCase):
|
|||||||
self.assertIn('configured-alias',result['title'])
|
self.assertIn('configured-alias',result['title'])
|
||||||
setup=receive('Details\n=======\nVMID Name Status Time Size Filename\n\nTotal running time: 0s\nTotal size: 0 B','error',subject.replace('multiple problems','unable to open storage'))
|
setup=receive('Details\n=======\nVMID Name Status Time Size Filename\n\nTotal running time: 0s\nTotal size: 0 B','error',subject.replace('multiple problems','unable to open storage'))
|
||||||
result=deliver(setup.event_type,setup.data,setup.severity)
|
result=deliver(setup.event_type,setup.data,setup.severity)
|
||||||
self.assertEqual(result['text'].count(setup.data['pve_title']),1)
|
self.assertEqual(result['text'].count('unable to open storage'),1)
|
||||||
|
self.assertNotIn(setup.data['pve_title'],result['text'])
|
||||||
unique=receive(NATIVE_REPORT,'error',subject.replace('multiple problems','job-end hook denied'))
|
unique=receive(NATIVE_REPORT,'error',subject.replace('multiple problems','job-end hook denied'))
|
||||||
result=deliver(unique.event_type,unique.data,unique.severity)
|
result=deliver(unique.event_type,unique.data,unique.severity)
|
||||||
self.assertIn('job-end hook denied',result['body'])
|
self.assertIn('job-end hook denied',result['body'])
|
||||||
|
|||||||
@@ -15,17 +15,7 @@ from notification_fixture import templates as actual_templates
|
|||||||
ROOT = Path(__file__).resolve().parents[3]
|
ROOT = Path(__file__).resolve().parents[3]
|
||||||
SCRIPTS = ROOT / 'AppImage/scripts'
|
SCRIPTS = ROOT / 'AppImage/scripts'
|
||||||
CATALOG = ROOT / 'AppImage/messages/en/common.json'
|
CATALOG = ROOT / 'AppImage/messages/en/common.json'
|
||||||
EXPECTED = {
|
EXPECTED = {'system_restore_completed': {'body': 'Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}\nThe node is now fully ready to use.'}}
|
||||||
'error_resolved': {
|
|
||||||
'title': '{hostname}: No longer reported - {category}{entity_suffix}',
|
|
||||||
'body': 'The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}',
|
|
||||||
'label': 'Recovery notification',
|
|
||||||
},
|
|
||||||
|
|
||||||
'system_restore_completed': {
|
|
||||||
'body': 'Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}',
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
def extract(path, name, owner=None, namespace=None):
|
def extract(path, name, owner=None, namespace=None):
|
||||||
@@ -82,7 +72,7 @@ class OutcomeWording(unittest.TestCase):
|
|||||||
result = module.render_template('system_restore_completed', {
|
result = module.render_template('system_restore_completed', {
|
||||||
'hostname':'node-a','guests':3,'stubs':0,'stale_nodes':0,
|
'hostname':'node-a','guests':3,'stubs':0,'stale_nodes':0,
|
||||||
'components':1,'duration':'2m','warnings_block':''}, 'es')
|
'components':1,'duration':'2m','warnings_block':''}, 'es')
|
||||||
self.assertIn('Configuraciones de guests aplicadas: 3', result['body'])
|
self.assertIn('Guests aplicados: 3', result['body'])
|
||||||
self.assertNotIn('invitados', result['body'].lower())
|
self.assertNotIn('invitados', result['body'].lower())
|
||||||
|
|
||||||
def test_settings_labels_stay_at_upstream_values_in_all_locales(self):
|
def test_settings_labels_stay_at_upstream_values_in_all_locales(self):
|
||||||
@@ -298,11 +288,6 @@ class OutcomeWording(unittest.TestCase):
|
|||||||
_SEV_DEFAULT = EmailChannel._SEV_DEFAULT
|
_SEV_DEFAULT = EmailChannel._SEV_DEFAULT
|
||||||
subject_prefix = 'ProxMenux'
|
subject_prefix = 'ProxMenux'
|
||||||
_build_detail_rows = staticmethod(build)
|
_build_detail_rows = staticmethod(build)
|
||||||
badge = catalog['channels']['email']['severity'].get('observation') or english['channels']['email']['severity']['observation']
|
|
||||||
recovery = fmt(Email(), 'No longer reported', 'Body', 'OK', {'_event_type': 'error_resolved',
|
|
||||||
'_notification_language': lang, '_group': 'health'})
|
|
||||||
self.assertIn('>' + badge.upper() + '</span>', recovery)
|
|
||||||
self.assertIn('color:#6b7280;', recovery)
|
|
||||||
unrelated = fmt(Email(), 'Reconnected', 'Body', 'OK', {'_event_type': 'node_reconnect',
|
unrelated = fmt(Email(), 'Reconnected', 'Body', 'OK', {'_event_type': 'node_reconnect',
|
||||||
'_notification_language': lang, '_group': 'cluster'})
|
'_notification_language': lang, '_group': 'cluster'})
|
||||||
self.assertIn('>' + catalog['channels']['email']['severity']['ok'].upper() + '</span>', unrelated)
|
self.assertIn('>' + catalog['channels']['email']['severity']['ok'].upper() + '</span>', unrelated)
|
||||||
@@ -346,7 +331,7 @@ class OutcomeWording(unittest.TestCase):
|
|||||||
self.catalog['runtime']['notifications']['backup'][key]).format(hostname=data['hostname'])
|
self.catalog['runtime']['notifications']['backup'][key]).format(hostname=data['hostname'])
|
||||||
self.assertTrue(result['title'].startswith(expected_title), result['title'])
|
self.assertTrue(result['title'].startswith(expected_title), result['title'])
|
||||||
else:
|
else:
|
||||||
self.assertEqual(result['title'], catalog['templates']['backup_complete']['title'].format_map(module._SafeFormatDict(data)))
|
self.assertEqual(result['title'], module.runtime_message('backup.unconfirmedTitle', lang, hostname=data['hostname']))
|
||||||
self.assertNotIn('{hostname}', result['title'])
|
self.assertNotIn('{hostname}', result['title'])
|
||||||
if state == 'unconfirmed':
|
if state == 'unconfirmed':
|
||||||
source = (catalog if catalog.get('backup', {}).get('unconfirmedBody')
|
source = (catalog if catalog.get('backup', {}).get('unconfirmedBody')
|
||||||
@@ -357,54 +342,12 @@ class OutcomeWording(unittest.TestCase):
|
|||||||
self.catalog['runtime']['notifications']['backup']['errorBody'], result['body'])
|
self.catalog['runtime']['notifications']['backup']['errorBody'], result['body'])
|
||||||
enriched, _ = module.enrich_with_emojis('backup_complete', result['title'], result['body'], data)
|
enriched, _ = module.enrich_with_emojis('backup_complete', result['title'], result['body'], data)
|
||||||
self.assertTrue(enriched.startswith({'confirmed':'💾✅','unconfirmed':'💾❔','failed':'💾❌'}[state]))
|
self.assertTrue(enriched.startswith({'confirmed':'💾✅','unconfirmed':'💾❔','failed':'💾❌'}[state]))
|
||||||
recovery = module.render_template('error_resolved', {'hostname':'node','category':'temperature',
|
|
||||||
'reason':'Old observation','duration':'3d','original_severity':'WARNING'}, lang)
|
|
||||||
recovery_source = catalog
|
|
||||||
self.assertEqual(recovery['title'], recovery_source['templates']['error_resolved']['title'].format(hostname='node',category='temperature',entity_suffix=''))
|
|
||||||
self.assertNotIn('resolved', recovery['title'].lower()) if lang == 'en' else None
|
|
||||||
restore = module.render_template('system_restore_completed', {'hostname':'node', 'guests':4,
|
restore = module.render_template('system_restore_completed', {'hostname':'node', 'guests':4,
|
||||||
'stubs':1,'stale_nodes':2,'components':1,'duration':'2m','warnings_block':'Missing module'},lang)
|
'stubs':1,'stale_nodes':2,'components':1,'duration':'2m','warnings_block':'Missing module'},lang)
|
||||||
self.assertIn('Missing module',restore['body'])
|
self.assertIn('Missing module',restore['body'])
|
||||||
self.assertNotIn('fully ready',restore['body'].lower())
|
if lang == 'en': self.assertIn('fully ready',restore['body'].lower())
|
||||||
|
|
||||||
def test_stale_record_disappearance_is_not_claimed_recovery(self):
|
|
||||||
data = {'hostname': 'node-a', 'category': 'temperature', 'reason': 'Temperature observation (no longer reported)',
|
|
||||||
'original_severity': 'WARNING', 'duration': '2d 0h', 'severity': 'OK'}
|
|
||||||
output = self.render('error_resolved', data, 'en')
|
|
||||||
self.assertIn('no longer in active health records', output['body'])
|
|
||||||
self.assertNotIn('resolved', (output['title'] + output['body']).lower())
|
|
||||||
self.assertIn('Time since first observation', output['body'])
|
|
||||||
|
|
||||||
def test_actual_poller_stale_disappearance_keeps_reason_factual(self):
|
|
||||||
class Store:
|
|
||||||
def get_active_errors(self): return []
|
|
||||||
def is_error_acknowledged(self, key): return False
|
|
||||||
class Event:
|
|
||||||
def __init__(self, *args, **kwargs): self.kind, self.severity, self.data = args[:3]
|
|
||||||
class Queue:
|
|
||||||
def __init__(self): self.items = []
|
|
||||||
def put(self, event): self.items.append(event)
|
|
||||||
ns = {'time': time, 'json': json, 'NotificationEvent': Event, 'Dict': dict}
|
|
||||||
poll = extract(SCRIPTS / 'notification_events.py', '_check_persistent_health', 'PollingCollector', ns)
|
|
||||||
class Collector:
|
|
||||||
_hostname = 'node-a'
|
|
||||||
_ENTITY_MAP = {'temperature': ('node', '')}
|
|
||||||
_first_poll_done = True
|
|
||||||
_known_errors = {'temp': {'category': 'temperature', 'reason': 'Temperature high',
|
|
||||||
'severity': 'WARNING', 'first_seen': '2026-09-25T00:00:00'}}
|
|
||||||
_notified_severity = {'temp': 'WARNING'}
|
|
||||||
_last_notified = {'temp': 1}
|
|
||||||
_queue = Queue()
|
|
||||||
def _guest_storage_error_is_now_foreign(self, *a): return False
|
|
||||||
def _save_known_errors_meta(self): pass
|
|
||||||
with patch.dict(sys.modules, {'health_persistence': types.SimpleNamespace(health_persistence=Store())}):
|
|
||||||
poll(Collector())
|
|
||||||
events = Collector._queue.items
|
|
||||||
self.assertEqual(len(events), 1)
|
|
||||||
self.assertEqual((events[0].kind, events[0].severity), ('error_resolved', 'OK'))
|
|
||||||
self.assertEqual(events[0].data['reason'], 'Temperature high (no longer reported)')
|
|
||||||
rendered = self.render(events[0].kind, events[0].data, 'en')
|
|
||||||
self.assertNotIn('recovered', rendered['body'].lower())
|
|
||||||
|
|
||||||
def test_warning_and_clean_restore_keep_only_reported_outcome(self):
|
def test_warning_and_clean_restore_keep_only_reported_outcome(self):
|
||||||
for warnings in ('', '⚠️ Boot sanity: missing modules\n'):
|
for warnings in ('', '⚠️ Boot sanity: missing modules\n'):
|
||||||
@@ -423,24 +366,9 @@ class OutcomeWording(unittest.TestCase):
|
|||||||
self.assertEqual(event['severity'], 'WARNING' if warnings else 'INFO')
|
self.assertEqual(event['severity'], 'WARNING' if warnings else 'INFO')
|
||||||
result = self.render(event['event_type'], event['data'], 'en')
|
result = self.render(event['event_type'], event['data'], 'en')
|
||||||
self.assertIn('Post-restore tasks completed', result['body'])
|
self.assertIn('Post-restore tasks completed', result['body'])
|
||||||
self.assertNotIn('fully ready', result['body'])
|
self.assertIn('fully ready', result['body'])
|
||||||
if warnings: self.assertIn('missing modules', result['body'])
|
if warnings: self.assertIn('missing modules', result['body'])
|
||||||
|
|
||||||
def test_missing_key_fallback_and_synthetic_translation(self):
|
|
||||||
catalog = copy.deepcopy(self.catalog)
|
|
||||||
translated = copy.deepcopy(self.catalog)
|
|
||||||
for event, fields in EXPECTED.items():
|
|
||||||
for field in fields: translated['runtime']['notifications']['templates'][event].pop(field)
|
|
||||||
_, render = renderer(catalog, translated)
|
|
||||||
for event, fields in EXPECTED.items():
|
|
||||||
for field in fields:
|
|
||||||
if field not in ('title', 'body'):
|
|
||||||
continue
|
|
||||||
self.assertEqual(render(event, {'category': 'disk'}, 'it')[field],
|
|
||||||
render(event, {'category': 'disk'}, 'en')[field])
|
|
||||||
translated['runtime']['notifications']['templates']['error_resolved']['title'] = 'Synthetic observation: {category}'
|
|
||||||
_, render = renderer(catalog, translated)
|
|
||||||
self.assertEqual(render('error_resolved', {'category': 'disk'}, 'it')['title'], 'Synthetic observation: disk')
|
|
||||||
|
|
||||||
def test_rich_backup_icon_tracks_outcome_and_digest_default_is_neutral(self):
|
def test_rich_backup_icon_tracks_outcome_and_digest_default_is_neutral(self):
|
||||||
tree = ast.parse((SCRIPTS / 'notification_templates.py').read_text())
|
tree = ast.parse((SCRIPTS / 'notification_templates.py').read_text())
|
||||||
@@ -465,8 +393,6 @@ class OutcomeWording(unittest.TestCase):
|
|||||||
tree = ast.parse((SCRIPTS / 'notification_templates.py').read_text())
|
tree = ast.parse((SCRIPTS / 'notification_templates.py').read_text())
|
||||||
icon_map = ast.literal_eval(next(n.value for n in tree.body if isinstance(n, ast.Assign)
|
icon_map = ast.literal_eval(next(n.value for n in tree.body if isinstance(n, ast.Assign)
|
||||||
and any(isinstance(t, ast.Name) and t.id == 'EVENT_EMOJI' for t in n.targets)))
|
and any(isinstance(t, ast.Name) and t.id == 'EVENT_EMOJI' for t in n.targets)))
|
||||||
for event in ('error_resolved', 'system_restore_completed'):
|
|
||||||
self.assertNotIn('✅', icon_map[event], event)
|
|
||||||
self.assertNotIn('✅', icon_map['backup_complete']) # buffered digest has no outcome metadata
|
self.assertNotIn('✅', icon_map['backup_complete']) # buffered digest has no outcome metadata
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,83 +0,0 @@
|
|||||||
"""Execute the build's literal backend copies, then a clean shipped-only runtime."""
|
|
||||||
import os
|
|
||||||
from pathlib import Path
|
|
||||||
import re
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
import tempfile
|
|
||||||
import unittest
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[3]
|
|
||||||
PROBE = r'''
|
|
||||||
import ast, pathlib, sys, types, typing
|
|
||||||
stage = pathlib.Path(sys.argv[1])
|
|
||||||
sys.path.insert(0, str(stage))
|
|
||||||
assert 'health_recovery' not in sys.modules
|
|
||||||
import notification_templates as templates
|
|
||||||
from notification_channels import EmailChannel
|
|
||||||
import health_recovery
|
|
||||||
for module in (templates, health_recovery, sys.modules['notification_channels']):
|
|
||||||
assert pathlib.Path(module.__file__).parent == stage
|
|
||||||
assert not any('projects/proxmox' in p for p in sys.path)
|
|
||||||
templates._get_hostname = lambda: 'node-a'
|
|
||||||
channel = object.__new__(EmailChannel)
|
|
||||||
channel.subject_prefix = '[ProxMenux]'
|
|
||||||
neutral = {'hostname': 'node-a', 'category': 'cpu', 'reason': 'CPU high', '_event_type': 'error_resolved'}
|
|
||||||
templates.render_template('error_resolved', neutral)
|
|
||||||
templates.enrich_with_emojis('error_resolved', 'Observation', 'Body', neutral)
|
|
||||||
channel._format_html('Observation', 'Body', 'OK', neutral)
|
|
||||||
templates.render_template('node_reconnect', {'hostname': 'node-a'})
|
|
||||||
now = 1000.0
|
|
||||||
calls = []
|
|
||||||
ns = dict(vars(typing), time=types.SimpleNamespace(time=lambda: now),
|
|
||||||
os=types.SimpleNamespace(cpu_count=lambda: 4),
|
|
||||||
psutil=types.SimpleNamespace(cpu_percent=lambda **kw: 20, cpu_count=lambda: 4),
|
|
||||||
health_persistence=types.SimpleNamespace(resolve_error=lambda *a, **kw: calls.append((a, kw))))
|
|
||||||
tree = ast.parse((stage/'health_monitor.py').read_text())
|
|
||||||
owner = next(n for n in tree.body if isinstance(n, ast.ClassDef) and n.name == 'HealthMonitor')
|
|
||||||
node = next(n for n in owner.body if isinstance(n, ast.FunctionDef) and n.name == '_check_cpu_with_hysteresis')
|
|
||||||
exec(compile(ast.Module(body=[node], type_ignores=[]), 'shipped_cpu', 'exec'), ns)
|
|
||||||
monitor = types.SimpleNamespace(state_history={'cpu_usage': [{'value':20,'time':now-i*10} for i in range(1,11)]},
|
|
||||||
CPU_WARNING=85, CPU_CRITICAL=95, CPU_RECOVERY=75, CPU_WARNING_DURATION=300,
|
|
||||||
CPU_CRITICAL_DURATION=300, CPU_RECOVERY_DURATION=120, _check_cpu_temperature=lambda:None)
|
|
||||||
assert ns['_check_cpu_with_hysteresis'](monitor)['status'] == 'OK'
|
|
||||||
assert len(calls) == 1 and calls[0][1]['check_evidence']['value'] == 20
|
|
||||||
proof = calls[0][1]['check_evidence']
|
|
||||||
native = dict(neutral, error_key='cpu_usage', check_evidence=proof, is_recovery=True, recovery_outcome='resolved')
|
|
||||||
# Actual body/icon/email consumers from the package, no source-module fixture rescue.
|
|
||||||
import unittest.mock
|
|
||||||
with unittest.mock.patch('health_recovery.time.time', return_value=now):
|
|
||||||
result = templates.render_template('error_resolved', native)
|
|
||||||
title, body = result['title'], result['body']
|
|
||||||
assert 'Resolved' in title and 'fresh health check' in body, (title, body, proof)
|
|
||||||
rich_title, rich_body = templates.enrich_with_emojis('error_resolved', title, body, native)
|
|
||||||
assert rich_title.startswith('✅')
|
|
||||||
assert 'background:#f0fdf4;' in channel._format_html(title, body, 'OK', native)
|
|
||||||
print('shipped-only neutral body/icon/email + CPU native measurement/proof consumers PASS')
|
|
||||||
'''
|
|
||||||
|
|
||||||
|
|
||||||
class PackagedRecoveryTests(unittest.TestCase):
|
|
||||||
def test_actual_copy_manifest_supports_isolated_recovery_runtime(self):
|
|
||||||
source = ROOT / 'AppImage/scripts'
|
|
||||||
build = (source / 'build_appimage.sh').read_text()
|
|
||||||
lines = [line for line in build.splitlines()
|
|
||||||
if re.match(r'^cp "\$SCRIPT_DIR/[^"/]+\.py" "\$APP_DIR/usr/bin/"', line)]
|
|
||||||
self.assertTrue(lines)
|
|
||||||
catalog_copy = re.search(r'^for locale in en de es fr it pt sk sv; do\n.*?^done$', build, re.MULTILINE | re.DOTALL)
|
|
||||||
if catalog_copy is None:
|
|
||||||
self.fail('The shipped locale-copy loop was not found in the actual build script')
|
|
||||||
lines.append(catalog_copy.group(0))
|
|
||||||
with tempfile.TemporaryDirectory(prefix='shipped-recovery-') as directory:
|
|
||||||
stage = Path(directory) / 'usr/bin'
|
|
||||||
stage.mkdir(parents=True)
|
|
||||||
copied = subprocess.run(['/bin/bash'], input='set -e\n'+'\n'.join(lines)+'\n', text=True,
|
|
||||||
capture_output=True, env={**os.environ, 'SCRIPT_DIR': str(source), 'APPIMAGE_ROOT': str(source.parent), 'APP_DIR': directory})
|
|
||||||
self.assertEqual(copied.returncode, 0, copied.stderr)
|
|
||||||
result = subprocess.run([sys.executable, '-I', '-B', '-c', PROBE, str(stage)],
|
|
||||||
cwd=directory, text=True, capture_output=True)
|
|
||||||
self.assertEqual(result.returncode, 0, result.stdout + result.stderr)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
unittest.main()
|
|
||||||
@@ -103,9 +103,7 @@ class PVE92Tests(unittest.TestCase):
|
|||||||
'warnings_block': ''}
|
'warnings_block': ''}
|
||||||
slovak = templates._load_runtime_catalog('sk')
|
slovak = templates._load_runtime_catalog('sk')
|
||||||
english = templates._load_runtime_catalog('en')
|
english = templates._load_runtime_catalog('en')
|
||||||
for event, field in (('error_resolved', 'title'), ('error_resolved', 'body'),
|
for event, field in (('system_restore_completed', 'body'),):
|
||||||
('system_restore_completed', 'body'),
|
|
||||||
('backup_complete', 'title'), ('backup_complete', 'body')):
|
|
||||||
with self.subTest(event=event, field=field):
|
with self.subTest(event=event, field=field):
|
||||||
value = slovak['templates'][event][field]
|
value = slovak['templates'][event][field]
|
||||||
result = templates.render_template(event, data, 'sk')
|
result = templates.render_template(event, data, 'sk')
|
||||||
|
|||||||
@@ -1,225 +0,0 @@
|
|||||||
"""Native initializer, measurement methods, SQL writers/readers and collector."""
|
|
||||||
import unittest
|
|
||||||
from notification_recovery_fixture import case, Clock, BASE, cpu, sql, poll
|
|
||||||
|
|
||||||
class RecoveryCorrectionTests(unittest.TestCase):
|
|
||||||
def test_supported_low_warning_current_violation_is_neutral(self):
|
|
||||||
with case() as store:
|
|
||||||
Clock.epoch = BASE - 600
|
|
||||||
initial = cpu(store, 80, [{'value':80, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=50)
|
|
||||||
self.assertEqual(initial['status'], 'WARNING')
|
|
||||||
first = sql(store, 'SELECT first_seen FROM errors')[0][0]
|
|
||||||
Clock.epoch = BASE
|
|
||||||
result = cpu(store, 60, [{'value':60, 'time':BASE-i*5} for i in range(1,26)], warning=50)
|
|
||||||
events, _ = poll(store, first, reason=initial['reason'])
|
|
||||||
# Preserve operational clear/hysteresis behavior, not its factual claim.
|
|
||||||
self.assertEqual(result['status'], 'OK')
|
|
||||||
self.assertFalse(events[0]['is_recovery'], events)
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage', first))
|
|
||||||
|
|
||||||
def test_original_policy_survives_repeated_native_updates(self):
|
|
||||||
import json
|
|
||||||
for later_warning in (85, 40):
|
|
||||||
with self.subTest(later_warning=later_warning), case() as store:
|
|
||||||
Clock.epoch = BASE - 600
|
|
||||||
cpu(store, 80, [{'value':80, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=50)
|
|
||||||
first = sql(store, 'SELECT first_seen FROM errors')[0][0]
|
|
||||||
for step in (400, 200):
|
|
||||||
Clock.epoch = BASE-step
|
|
||||||
cpu(store, 90, [{'value':90, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=later_warning)
|
|
||||||
original = json.loads(sql(store, 'SELECT details FROM errors')[0][0])['cpu_policy']
|
|
||||||
self.assertEqual(original['warning'], 50)
|
|
||||||
Clock.epoch = BASE
|
|
||||||
cpu(store, 20, warning=later_warning)
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage', first))
|
|
||||||
self.assertFalse(poll(store, first)[0][0]['is_recovery'])
|
|
||||||
|
|
||||||
def test_clock_rollback_new_row_generic_clear_cannot_inherit_old_proof(self):
|
|
||||||
for rollback, reuse_first in ((True,False),(False,False),(True,True)):
|
|
||||||
with self.subTest(rollback=rollback,reuse_first=reuse_first),case() as store:
|
|
||||||
Clock.epoch = BASE-600
|
|
||||||
cpu(store, 90, [{'value':90, 'time':Clock.epoch-i*5} for i in range(1,26)])
|
|
||||||
first = sql(store, 'SELECT first_seen FROM errors')[0][0]
|
|
||||||
Clock.epoch = BASE; Clock.tick = .001
|
|
||||||
try: cpu(store)
|
|
||||||
finally: Clock.tick = 0
|
|
||||||
self.assertTrue(store.get_recovery_evidence('cpu_usage', first))
|
|
||||||
store.acknowledge_error('cpu_usage', suppression_hours=-1)
|
|
||||||
store.clear_error('cpu_usage')
|
|
||||||
Clock.epoch = BASE-100 if rollback else BASE+1
|
|
||||||
store.record_error('cpu_usage','cpu','WARNING','new incident after clock step',{})
|
|
||||||
second = sql(store, 'SELECT first_seen FROM errors')[0][0]
|
|
||||||
self.assertNotEqual(first, second)
|
|
||||||
if reuse_first:
|
|
||||||
# Restored malformed snapshot with reused wall-clock identity;
|
|
||||||
# native row id and latest closure still prevent replay.
|
|
||||||
sql(store,'UPDATE errors SET first_seen=?',(first,));second=first
|
|
||||||
Clock.epoch = BASE+.0005 if rollback else BASE+2
|
|
||||||
store.clear_error('cpu_usage'); Clock.epoch = BASE+3
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage', second))
|
|
||||||
self.assertFalse(poll(store, second)[0][0]['is_recovery'])
|
|
||||||
|
|
||||||
def test_malformed_native_and_manual_proof_is_neutral_at_all_consumers(self):
|
|
||||||
import copy, json, time
|
|
||||||
from unittest.mock import patch
|
|
||||||
from notification_fixture import LANGUAGES
|
|
||||||
from notification_final_fixture import deliver
|
|
||||||
with case() as store:
|
|
||||||
Clock.epoch = BASE-600
|
|
||||||
cpu(store, 90, [{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)])
|
|
||||||
first = sql(store,'SELECT first_seen FROM errors')[0][0]
|
|
||||||
Clock.epoch = BASE; cpu(store)
|
|
||||||
event_data = json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0])
|
|
||||||
data = poll(store,first)[0][0]
|
|
||||||
for field, bad in [('checked_at','bad'),('checked_at',True),('checked_at',float('inf')),
|
|
||||||
('checked_at',10**400),('value',60),('max_sample',float('nan')),
|
|
||||||
('normal_samples',True),('normal_samples',9),('checked_at',BASE+1),('checked_at',BASE-7201),
|
|
||||||
('policy',{'warning':False,'critical':95,'recovery':75}),
|
|
||||||
('policy',{'warning':96,'critical':95,'recovery':75}),
|
|
||||||
('policy',{'warning':85,'critical':95}),
|
|
||||||
('policy',{'warning':85,'critical':95,'recovery':float('nan')}),
|
|
||||||
('policy',{'warning':85,'critical':95,'recovery':75,'extra':0})]:
|
|
||||||
broken = copy.deepcopy(event_data)
|
|
||||||
broken['check_evidence'][field] = bad
|
|
||||||
if field == 'value': broken['check_evidence']['policy']['warning'] = 50
|
|
||||||
sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps(broken),))
|
|
||||||
with self.subTest(field=field,bad=str(bad)),patch('health_recovery.time.time',return_value=BASE):
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage',first))
|
|
||||||
for language in LANGUAGES:
|
|
||||||
for manual in (False,True):
|
|
||||||
result = deliver('error_resolved',{**data,'check_evidence':broken['check_evidence']},'OK',language,manual=manual)
|
|
||||||
self.assertNotIn('background:#f0fdf4;', result['html'])
|
|
||||||
# Caller content is trusted, not authenticated native proof; even
|
|
||||||
# well-shaped assertions require a valid time/type/numeric contract.
|
|
||||||
for proof in ({'check':'cpu_usage'}, {'check':'cpu_usage','checked_at':time.time()+1}):
|
|
||||||
self.assertNotIn('background:#f0fdf4;', deliver('error_resolved',{**data,'check_evidence':proof},'OK',manual=True)['html'])
|
|
||||||
|
|
||||||
def test_exact_service_active_native_clear_reaches_recovery_consumers(self):
|
|
||||||
from unittest.mock import patch
|
|
||||||
from notification_recovery_fixture import service
|
|
||||||
from notification_fixture import LANGUAGES
|
|
||||||
from notification_final_fixture import deliver
|
|
||||||
with case() as store:
|
|
||||||
Clock.epoch = BASE-600
|
|
||||||
service(store,3,'inactive\n')
|
|
||||||
first = sql(store,'SELECT first_seen FROM errors')[0][0]
|
|
||||||
Clock.epoch = BASE
|
|
||||||
result, calls = service(store)
|
|
||||||
self.assertEqual(result['status'],'OK')
|
|
||||||
self.assertEqual(calls,[(['systemctl','is-active','pvedaemon'], {'capture_output':True,'text':True,'timeout':2})])
|
|
||||||
proof = store.get_recovery_evidence('pve_service_pvedaemon',first)
|
|
||||||
self.assertTrue(proof)
|
|
||||||
data = poll(store,first,'pve_service_pvedaemon','pve_services','PVE service pvedaemon is inactive',details={'service':'pvedaemon'})[0][0]
|
|
||||||
self.assertTrue(data['is_recovery'])
|
|
||||||
with patch('health_recovery.time.time',return_value=BASE):
|
|
||||||
for language in LANGUAGES:
|
|
||||||
for manual in (False,True):
|
|
||||||
rendered = deliver('error_resolved',data,'OK',language,manual=manual)
|
|
||||||
self.assertIn('background:#f0fdf4;',rendered['html'])
|
|
||||||
self.assertIn('pvedaemon', rendered['text'])
|
|
||||||
|
|
||||||
def test_cpu_positive_default_low_policy_and_neutral_history_controls(self):
|
|
||||||
from notification_recovery_fixture import record
|
|
||||||
for warning,current,history,legacy,expected in (
|
|
||||||
(85,20,None,False,True), (50,20,None,False,True),
|
|
||||||
(85,20,None,True,False), (85,99,[],False,False),
|
|
||||||
(85,20,[],False,False),
|
|
||||||
(85,20,[{'value':20,'time':BASE+i*5} for i in range(1,10)],False,False),
|
|
||||||
(85,20,[{'value':20,'time':BASE-121-i} for i in range(12)],False,False),
|
|
||||||
(85,99,None,False,False),
|
|
||||||
(85,float('nan'),None,False,False), (85,float('inf'),None,False,False),
|
|
||||||
(85,True,None,False,False), (85,10**400,None,False,False)):
|
|
||||||
with self.subTest(warning=warning,current=str(current),legacy=legacy), case() as store:
|
|
||||||
if legacy: first = record(store)
|
|
||||||
else:
|
|
||||||
Clock.epoch=BASE-600
|
|
||||||
cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)],warning=warning)
|
|
||||||
first=sql(store,'SELECT first_seen FROM errors')[0][0]
|
|
||||||
Clock.epoch=BASE; cpu(store,current,history,warning=warning)
|
|
||||||
self.assertEqual(bool(store.get_recovery_evidence('cpu_usage',first)),expected)
|
|
||||||
|
|
||||||
def test_service_unavailable_removed_overall_ok_and_ack_controls(self):
|
|
||||||
from notification_recovery_fixture import service, record
|
|
||||||
for rc,stdout,raised in ((3,'inactive\n',False),(4,'unknown\n',False),(0,'active extra\n',False),(1,'active\n',False),(0,'',True)):
|
|
||||||
with self.subTest(rc=rc,stdout=stdout,raised=raised),case() as store:
|
|
||||||
Clock.epoch=BASE-600; service(store,3,'inactive\n')
|
|
||||||
first=sql(store,'SELECT first_seen FROM errors')[0][0]
|
|
||||||
Clock.epoch=BASE; service(store,rc,stdout,raised)
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('pve_service_pvedaemon',first))
|
|
||||||
self.assertFalse(poll(store,first,'pve_service_pvedaemon','pve_services')[0])
|
|
||||||
for removed in ('pvedaemon','corosync'):
|
|
||||||
with self.subTest(removed=removed),case() as store:
|
|
||||||
first=record(store,'pve_service_'+removed,'pve_services','service inactive',{'service':removed})
|
|
||||||
result,calls=service(store,services=(),clustered=False)
|
|
||||||
self.assertEqual(result['status'],'OK'); self.assertEqual(calls,[])
|
|
||||||
store.clear_error('pve_service_'+removed)
|
|
||||||
self.assertFalse(poll(store,first,'pve_service_'+removed,'pve_services')[0][0]['is_recovery'])
|
|
||||||
with case() as store:
|
|
||||||
first=record(store,'pve_service_corosync','pve_services','corosync inactive',{'service':'corosync'})
|
|
||||||
result,calls=service(store,services=('pvedaemon',),clustered=False)
|
|
||||||
self.assertEqual(result['status'],'OK')
|
|
||||||
self.assertEqual([c[0][-1] for c in calls],['pvedaemon'])
|
|
||||||
self.assertTrue(store.is_error_active('pve_service_corosync'))
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('pve_service_corosync',first))
|
|
||||||
with case() as store:
|
|
||||||
first=record(store,'pve_service_pvedaemon','pve_services','service inactive')
|
|
||||||
store.acknowledge_error('pve_service_pvedaemon',suppression_hours=-1)
|
|
||||||
store.clear_error('pve_service_pvedaemon',check_evidence={'check':'pve_service_pvedaemon','checked_at':BASE,'service':'pvedaemon','state':'active','returncode':0})
|
|
||||||
self.assertEqual(sql(store,'SELECT id FROM errors'),[])
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('pve_service_pvedaemon',first))
|
|
||||||
|
|
||||||
def test_native_binding_latest_closure_rollbacks_and_consistent_ack_read(self):
|
|
||||||
import contextlib, json, sqlite3
|
|
||||||
for mutation in ('row_id','first_seen','closure','last_seen','latest_clear','latest_resolve','ack','event_insert_failure','ack_before_join'):
|
|
||||||
with self.subTest(mutation=mutation),case() as store:
|
|
||||||
Clock.epoch=BASE-600
|
|
||||||
cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)])
|
|
||||||
first=sql(store,'SELECT first_seen FROM errors')[0][0]
|
|
||||||
Clock.epoch=BASE
|
|
||||||
if mutation=='event_insert_failure':
|
|
||||||
sql(store,"CREATE TRIGGER fail_resolve BEFORE INSERT ON events WHEN NEW.event_type='resolved' BEGIN SELECT RAISE(ABORT,'fixture'); END")
|
|
||||||
self.assertEqual(cpu(store)['status'],'UNKNOWN')
|
|
||||||
self.assertIsNone(sql(store,'SELECT resolved_at FROM errors')[0][0])
|
|
||||||
continue
|
|
||||||
cpu(store); self.assertTrue(store.get_recovery_evidence('cpu_usage',first))
|
|
||||||
if mutation in ('row_id','first_seen','closure'):
|
|
||||||
data=json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0])
|
|
||||||
field={'row_id':'id','first_seen':'first_seen','closure':'resolved_at'}[mutation]
|
|
||||||
data['incident'][field]='wrong'
|
|
||||||
sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps(data),))
|
|
||||||
elif mutation=='last_seen': sql(store,'UPDATE errors SET last_seen=?',(Clock.fromtimestamp(BASE+1).isoformat(),))
|
|
||||||
elif mutation.startswith('latest_'):
|
|
||||||
sql(store,"INSERT INTO events(event_type,error_key,timestamp,data) VALUES(?,'cpu_usage',?,'{}')",('cleared' if mutation=='latest_clear' else 'resolved',Clock.now().isoformat()))
|
|
||||||
elif mutation=='ack': store.acknowledge_error('cpu_usage',suppression_hours=-1)
|
|
||||||
else:
|
|
||||||
original=store._db_connection; triggered=[]
|
|
||||||
class Proxy:
|
|
||||||
def __init__(self,connection):self.connection=connection
|
|
||||||
def __getattr__(self,name):return getattr(self.connection,name)
|
|
||||||
def execute(self,query,args=()):
|
|
||||||
if 'FROM errors e JOIN events' in query and not triggered:
|
|
||||||
triggered.append(True)
|
|
||||||
store.acknowledge_error('cpu_usage',suppression_hours=-1)
|
|
||||||
return self.connection.execute(query,args)
|
|
||||||
@contextlib.contextmanager
|
|
||||||
def interleaved(**kw):
|
|
||||||
with original(**kw) as conn: yield Proxy(conn)
|
|
||||||
store._db_connection=interleaved
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage',first))
|
|
||||||
if mutation=='ack_before_join': self.assertTrue(triggered)
|
|
||||||
with case() as store:
|
|
||||||
sql(store,"INSERT INTO errors(error_key,category,severity,reason,first_seen,last_seen) VALUES('cpu_usage','cpu','WARNING','fixture','x','x')")
|
|
||||||
with self.assertRaises(sqlite3.IntegrityError):
|
|
||||||
sql(store,"INSERT INTO errors(error_key,category,severity,reason,first_seen,last_seen) VALUES('cpu_usage','cpu','WARNING','duplicate','x','x')")
|
|
||||||
|
|
||||||
def test_malformed_history_declines_proof_without_changing_operational_clear(self):
|
|
||||||
with case() as store:
|
|
||||||
Clock.epoch=BASE-600
|
|
||||||
cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)])
|
|
||||||
first=sql(store,'SELECT first_seen FROM errors')[0][0]
|
|
||||||
Clock.epoch=BASE
|
|
||||||
result=cpu(store,20,[{'value':10**400,'time':BASE-i*5} for i in range(1,10)])
|
|
||||||
self.assertEqual(result['status'],'OK')
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage',first))
|
|
||||||
|
|
||||||
if __name__ == '__main__': unittest.main()
|
|
||||||
@@ -1,75 +0,0 @@
|
|||||||
"""Fresh existing-check provenance; native initializer and disposable SQLite."""
|
|
||||||
import json
|
|
||||||
import time
|
|
||||||
import unittest
|
|
||||||
from unittest.mock import patch
|
|
||||||
from notification_fixture import templates, LANGUAGES
|
|
||||||
from notification_final_fixture import deliver
|
|
||||||
from notification_recovery_fixture import case, Clock, BASE, cpu, sql, poll
|
|
||||||
|
|
||||||
|
|
||||||
def original_cpu(store):
|
|
||||||
Clock.epoch = BASE-600
|
|
||||||
result = cpu(store, 90, [{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)])
|
|
||||||
assert result['status'] == 'WARNING'
|
|
||||||
Clock.epoch = BASE
|
|
||||||
return sql(store, 'SELECT first_seen FROM errors')[0][0]
|
|
||||||
|
|
||||||
|
|
||||||
class RecoveryEvidenceTests(unittest.TestCase):
|
|
||||||
def test_cpu_success_provenance_is_persisted_only_after_normal_samples(self):
|
|
||||||
with case() as store:
|
|
||||||
first = original_cpu(store)
|
|
||||||
self.assertEqual(cpu(store)['status'], 'OK')
|
|
||||||
proof = store.get_recovery_evidence('cpu_usage', first)
|
|
||||||
self.assertTrue(proof)
|
|
||||||
self.assertEqual(proof['check'], 'cpu_usage')
|
|
||||||
self.assertEqual(proof['checked_at'], BASE)
|
|
||||||
# A generic closure never gains proof; use another actual native row.
|
|
||||||
store.record_error('pve_service_test','pve_services','CRITICAL','inactive')
|
|
||||||
store.resolve_error('pve_service_test','No longer present')
|
|
||||||
self.assertFalse(json.loads(sql(store,"SELECT data FROM events ORDER BY id DESC LIMIT 1")[0][0]).get('check_evidence'))
|
|
||||||
|
|
||||||
def test_recovery_query_requires_fresh_same_incident_proof(self):
|
|
||||||
with case() as store:
|
|
||||||
first = original_cpu(store); cpu(store)
|
|
||||||
proof = store.get_recovery_evidence('cpu_usage',first)
|
|
||||||
self.assertTrue(proof)
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage','different incident'))
|
|
||||||
saved = sql(store,'SELECT last_seen,resolved_at FROM errors')[0]
|
|
||||||
for field,value in [('acknowledged',1),('resolved_at',None),('last_seen',Clock.fromtimestamp(BASE+1).isoformat())]:
|
|
||||||
sql(store,f'UPDATE errors SET {field}=?',(value,))
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage',first))
|
|
||||||
sql(store,'UPDATE errors SET acknowledged=0,last_seen=?,resolved_at=?',saved)
|
|
||||||
data = json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0])
|
|
||||||
for bad in (None,{'check':'cpu_usage','checked_at':BASE-7201}, {'check':'storage_removed','checked_at':BASE}, {'check':'cpu_usage','checked_at':float('inf')}, {'check':'cpu_usage','checked_at':10**400}):
|
|
||||||
sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps({**data,'check_evidence':bad}),))
|
|
||||||
self.assertIsNone(store.get_recovery_evidence('cpu_usage',first))
|
|
||||||
|
|
||||||
def test_poller_and_all_consumers_distinguish_proven_recovery_from_disappearance(self):
|
|
||||||
for proved in (False,True):
|
|
||||||
with case() as store:
|
|
||||||
first = original_cpu(store)
|
|
||||||
if proved: cpu(store)
|
|
||||||
else: store.resolve_error('cpu_usage','No longer present')
|
|
||||||
data = poll(store,first,reason='CPU high')[0][0]
|
|
||||||
self.assertEqual(data['is_recovery'],proved)
|
|
||||||
self.assertEqual(data['recovery_outcome'],'resolved' if proved else 'no_longer_reported')
|
|
||||||
with patch('health_recovery.time.time',return_value=BASE):
|
|
||||||
for lang in LANGUAGES:
|
|
||||||
for manual in (False,True):
|
|
||||||
result = deliver('error_resolved',data,'OK',lang,manual=manual)
|
|
||||||
self.assertEqual('background:#f0fdf4;' in result['html'],proved)
|
|
||||||
if proved:
|
|
||||||
self.assertIn(templates.runtime_message('healthRecovery.title',lang,hostname='node-a',category='cpu',entity_suffix=''),result['title'])
|
|
||||||
self.assertEqual(result['text'].count(data['reason']),1)
|
|
||||||
|
|
||||||
def test_manual_recovery_flag_alone_is_not_authoritative_evidence(self):
|
|
||||||
data={'hostname':'node-a','category':'cpu','reason':'Observation disappeared','duration':'1h','original_severity':'WARNING','recovery_outcome':'resolved'}
|
|
||||||
for lang in LANGUAGES:
|
|
||||||
for manual in (False,True):
|
|
||||||
result=deliver('error_resolved',data,'OK',lang,manual=manual)
|
|
||||||
self.assertNotIn('background:#f0fdf4;',result['html'])
|
|
||||||
self.assertNotIn(templates.runtime_message('healthRecovery.body',lang,**data),result['body'])
|
|
||||||
|
|
||||||
if __name__=='__main__':unittest.main()
|
|
||||||
@@ -1,170 +0,0 @@
|
|||||||
"""Durable native observation order, not wall-clock row identity, admits proof."""
|
|
||||||
import datetime
|
|
||||||
import json
|
|
||||||
import re
|
|
||||||
import sys
|
|
||||||
import types
|
|
||||||
import typing
|
|
||||||
import unittest
|
|
||||||
from unittest.mock import patch
|
|
||||||
from notification_recovery_fixture import case, Clock, BASE, cpu, service, sql, extract, scripts, TIME
|
|
||||||
from notification_fixture import LANGUAGES
|
|
||||||
from notification_final_fixture import deliver
|
|
||||||
|
|
||||||
|
|
||||||
def collector(store):
|
|
||||||
events = []
|
|
||||||
target = types.SimpleNamespace(_hostname='alias {rack.location}',
|
|
||||||
_ENTITY_MAP={'cpu':('node',''), 'pve_services':('node','')}, _first_poll_done=False,
|
|
||||||
_known_errors={}, _notified_severity={}, _last_notified={}, SAME_ERROR_COOLDOWN=86400,
|
|
||||||
_get_cooldown_from_db=lambda *a:BASE-1, _queue=types.SimpleNamespace(put=events.append),
|
|
||||||
_save_known_errors_meta=lambda:None)
|
|
||||||
ns = dict(vars(typing), time=TIME, json=json, re=re,
|
|
||||||
NotificationEvent=lambda *a, **kw:types.SimpleNamespace(event_type=a[0], severity=a[1], data=a[2], **kw),
|
|
||||||
startup_grace=types.SimpleNamespace(should_suppress_category=lambda *a:False))
|
|
||||||
target._guest_storage_error_is_now_foreign = extract(scripts/'notification_events.py',
|
|
||||||
'_guest_storage_error_is_now_foreign', 'PollingCollector', ns)
|
|
||||||
poll = extract(scripts/'notification_events.py', '_check_persistent_health', 'PollingCollector', ns)
|
|
||||||
def tick():
|
|
||||||
with patch.dict(sys.modules, {'health_persistence':types.SimpleNamespace(health_persistence=store),
|
|
||||||
'datetime':types.SimpleNamespace(**{**vars(datetime), 'datetime':Clock})}):
|
|
||||||
poll(target)
|
|
||||||
return target, events, tick
|
|
||||||
|
|
||||||
|
|
||||||
def abnormal(store, kind, value=99):
|
|
||||||
if kind == 'cpu':
|
|
||||||
return cpu(store, value, [{'value':value, 'time':Clock.epoch-i*5} for i in range(1,26)])
|
|
||||||
return service(store,3,'inactive\n')[0]
|
|
||||||
|
|
||||||
|
|
||||||
def normal(store, kind):
|
|
||||||
return cpu(store) if kind == 'cpu' else service(store)[0]
|
|
||||||
|
|
||||||
|
|
||||||
def initial_closure(store, kind):
|
|
||||||
key = 'cpu_usage' if kind == 'cpu' else 'pve_service_pvedaemon'
|
|
||||||
Clock.epoch = BASE-600
|
|
||||||
abnormal(store,kind,90)
|
|
||||||
target, events, tick = collector(store)
|
|
||||||
tick()
|
|
||||||
assert target._first_poll_done and key in target._known_errors and not events
|
|
||||||
snapshot = json.loads(json.dumps(target._known_errors))
|
|
||||||
Clock.epoch = BASE
|
|
||||||
assert normal(store,kind)['status'] == 'OK'
|
|
||||||
first = snapshot[key]['first_seen']
|
|
||||||
assert store.get_recovery_evidence(key,first)
|
|
||||||
return key, first, snapshot, target, events, tick
|
|
||||||
|
|
||||||
|
|
||||||
class RecoveryOrderTests(unittest.TestCase):
|
|
||||||
def assert_consumers(self, data, expected):
|
|
||||||
with patch('health_recovery.time.time', return_value=Clock.epoch):
|
|
||||||
for language in LANGUAGES:
|
|
||||||
for manual in (False,True):
|
|
||||||
with self.subTest(language=language, manual=manual):
|
|
||||||
result = deliver('error_resolved',data,'OK',language,manual=manual)
|
|
||||||
self.assertEqual('background:#f0fdf4;' in result['html'],expected)
|
|
||||||
self.assertIn('alias {rack.location}',result['text'])
|
|
||||||
quiet = deliver('error_resolved',data,'OK',language,quiet=True)
|
|
||||||
self.assertEqual(len(quiet['buffered']),1)
|
|
||||||
# The existing quiet digest stores the rendered title, not
|
|
||||||
# health body/proof metadata. Assert its exact outcome label.
|
|
||||||
from notification_fixture import templates
|
|
||||||
label = templates.render_template('error_resolved',data,language)['title'].split(': ',1)[-1]
|
|
||||||
self.assertIn(label, quiet['body'])
|
|
||||||
self.assertIn(label, quiet['buffered'][0][2])
|
|
||||||
|
|
||||||
def assert_superseded(self, kind):
|
|
||||||
for offset in (-100,0,1):
|
|
||||||
for value in ((90,99) if kind == 'cpu' else (99,)):
|
|
||||||
with self.subTest(kind=kind, offset=offset, value=value), case() as store:
|
|
||||||
key, first, snapshot, target, events, tick = initial_closure(store,kind)
|
|
||||||
oldrow = sql(store,'SELECT id,first_seen,resolved_at FROM errors')[0]
|
|
||||||
prior = sql(store,'SELECT id,event_type FROM events ORDER BY id')
|
|
||||||
Clock.epoch = BASE+offset
|
|
||||||
renewed = abnormal(store,kind,value)
|
|
||||||
self.assertEqual(renewed['status'],'WARNING' if kind == 'cpu' and value == 90 else 'CRITICAL')
|
|
||||||
self.assertEqual(sql(store,'SELECT id,first_seen,resolved_at FROM errors')[0],oldrow)
|
|
||||||
later = sql(store,'SELECT id,event_type FROM events ORDER BY id')[-1]
|
|
||||||
self.assertGreater(later[0],prior[-1][0])
|
|
||||||
self.assertEqual(later[1],'escalated' if kind == 'cpu' and value == 99 else 'updated')
|
|
||||||
if kind == 'cpu':
|
|
||||||
policy=json.loads(sql(store,'SELECT details FROM errors')[0][0])['cpu_policy']
|
|
||||||
self.assertEqual(policy,{'warning':85,'critical':95,'recovery':75})
|
|
||||||
Clock.epoch = BASE+2
|
|
||||||
tick()
|
|
||||||
self.assertEqual(len(events),1)
|
|
||||||
data = events[0].data
|
|
||||||
self.assertFalse(data['is_recovery'])
|
|
||||||
self.assertIsNone(store.get_recovery_evidence(key,first))
|
|
||||||
self.assert_consumers(data,False)
|
|
||||||
# Existing operations do not rearm the resolved row, so a
|
|
||||||
# later normal check cannot establish a NEW native closure.
|
|
||||||
before = sql(store,'SELECT id FROM events ORDER BY id')
|
|
||||||
Clock.epoch = BASE+3
|
|
||||||
self.assertEqual(normal(store,kind)['status'],'OK')
|
|
||||||
self.assertEqual(sql(store,'SELECT id FROM events ORDER BY id'),before)
|
|
||||||
self.assertIsNone(store.get_recovery_evidence(key,first))
|
|
||||||
|
|
||||||
def test_cpu_superseded_same_row_is_neutral_at_all_consumers(self):
|
|
||||||
self.assert_superseded('cpu')
|
|
||||||
|
|
||||||
def test_service_superseded_same_row_is_neutral_at_all_consumers(self):
|
|
||||||
self.assert_superseded('service')
|
|
||||||
|
|
||||||
def test_fresh_closure_and_repeated_noop_clear_keep_genuine_proof(self):
|
|
||||||
for kind in ('cpu','service'):
|
|
||||||
with self.subTest(kind=kind),case() as store:
|
|
||||||
key, first, snapshot, target, events, tick = initial_closure(store,kind)
|
|
||||||
before = sql(store,'SELECT id,event_type FROM events ORDER BY id')
|
|
||||||
Clock.epoch = BASE+1
|
|
||||||
for _ in range(3):
|
|
||||||
store.clear_error(key)
|
|
||||||
store.resolve_error(key,'generic repeat')
|
|
||||||
normal(store,kind)
|
|
||||||
self.assertEqual(sql(store,'SELECT id,event_type FROM events ORDER BY id'),before)
|
|
||||||
self.assertTrue(store.get_recovery_evidence(key,first))
|
|
||||||
tick()
|
|
||||||
self.assertEqual(len(events),1)
|
|
||||||
self.assertTrue(events[0].data['is_recovery'])
|
|
||||||
self.assert_consumers(events[0].data,True)
|
|
||||||
tick()
|
|
||||||
self.assertEqual(len(events),1)
|
|
||||||
|
|
||||||
def test_later_actual_generic_closure_blocks_older_proof_even_clock_rollback(self):
|
|
||||||
for kind in ('cpu','service'):
|
|
||||||
with self.subTest(kind=kind),case() as store:
|
|
||||||
key, first, *_ = initial_closure(store,kind)
|
|
||||||
Clock.epoch = BASE-100
|
|
||||||
with store._db_connection() as conn:
|
|
||||||
store._record_event(conn.cursor(),'cleared',key,{'reason':'generic actual closure','check_evidence':None})
|
|
||||||
conn.commit()
|
|
||||||
self.assertIsNone(store.get_recovery_evidence(key,first))
|
|
||||||
|
|
||||||
def test_explicit_acknowledged_row_suppresses_native_proof(self):
|
|
||||||
for kind in ('cpu','service'):
|
|
||||||
with self.subTest(kind=kind),case() as store:
|
|
||||||
key, first, snapshot, target, events, tick = initial_closure(store,kind)
|
|
||||||
store.acknowledge_error(key,suppression_hours=-1)
|
|
||||||
self.assertEqual(sql(store,'SELECT acknowledged FROM errors')[0][0],1)
|
|
||||||
self.assertIsNone(store.get_recovery_evidence(key,first))
|
|
||||||
tick()
|
|
||||||
self.assertEqual(events,[])
|
|
||||||
|
|
||||||
def test_new_incarnation_abnormal_order_blocks_previous_closure(self):
|
|
||||||
for kind in ('cpu','service'):
|
|
||||||
with self.subTest(kind=kind),case() as store:
|
|
||||||
key, first, *_ = initial_closure(store,kind)
|
|
||||||
old_id = sql(store,'SELECT id FROM errors')[0][0]
|
|
||||||
store.acknowledge_error(key,suppression_hours=-1)
|
|
||||||
store.clear_error(key)
|
|
||||||
Clock.epoch = BASE-100
|
|
||||||
abnormal(store,kind,99)
|
|
||||||
self.assertGreater(sql(store,'SELECT id FROM errors')[0][0],old_id)
|
|
||||||
self.assertEqual(sql(store,'SELECT event_type FROM events ORDER BY id DESC')[0][0],'new')
|
|
||||||
self.assertIsNone(store.get_recovery_evidence(key,first))
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
unittest.main()
|
|
||||||
@@ -6266,8 +6266,8 @@
|
|||||||
"label": "Neues Gesundheitsproblem"
|
"label": "Neues Gesundheitsproblem"
|
||||||
},
|
},
|
||||||
"error_resolved": {
|
"error_resolved": {
|
||||||
"title": "{hostname}: Nicht mehr gemeldet – {category}{entity_suffix}",
|
"title": "{hostname}: Gelöst – {category}{entity_suffix}",
|
||||||
"body": "Das Problem in der Kategorie {category} ist nicht mehr in den aktiven Zustandsmeldungen enthalten.\n{reason}\n🚦 Vorheriger Schweregrad: {original_severity}\n⏱️ Zeit seit der ersten Meldung: {duration}",
|
"body": "Das Problem {category} wurde behoben.\n{reason}\n🚦 Vorheriger Schweregrad: {original_severity}\n⏱️ Dauer: {duration}",
|
||||||
"label": "Wiederherstellungsbenachrichtigung"
|
"label": "Wiederherstellungsbenachrichtigung"
|
||||||
},
|
},
|
||||||
"error_escalated": {
|
"error_escalated": {
|
||||||
@@ -6396,8 +6396,8 @@
|
|||||||
"label": "Sicherung gestartet"
|
"label": "Sicherung gestartet"
|
||||||
},
|
},
|
||||||
"backup_complete": {
|
"backup_complete": {
|
||||||
"title": "{hostname}: Backup-Ergebnis nicht bestätigt",
|
"title": "{hostname} → {storage}: Sicherung abgeschlossen – {vmname} ({vmid})",
|
||||||
"body": "Das Backup-Ergebnis lässt sich anhand dieser Meldung nicht bestätigen.",
|
"body": "Die Sicherung von {vmname} (ID: {vmid}) wurde am {storage} erfolgreich abgeschlossen.\nGröße: {size}",
|
||||||
"label": "Sicherung abgeschlossen"
|
"label": "Sicherung abgeschlossen"
|
||||||
},
|
},
|
||||||
"backup_warning": {
|
"backup_warning": {
|
||||||
@@ -6557,7 +6557,7 @@
|
|||||||
},
|
},
|
||||||
"system_restore_completed": {
|
"system_restore_completed": {
|
||||||
"title": "{hostname}: Host-Wiederherstellung abgeschlossen",
|
"title": "{hostname}: Host-Wiederherstellung abgeschlossen",
|
||||||
"body": "Aufgaben nach der Wiederherstellung im Hintergrund abgeschlossen.\n\nÜbernommene Gäste: {guests}\nBind-Mount-Platzhalter: {stubs}\nEntfernte veraltete Knotenverzeichnisse: {stale_nodes}\nNeu installierte Komponenten: {components}\nDauer: {duration}\n{warnings_block}",
|
"body": "Aufgaben nach der Wiederherstellung im Hintergrund ausgeführt.\n\nGäste haben sich beworben: {guests}\nBind-Mount-Stubs: {stubs}\nVeraltete Knotenverzeichnisse entfernt: {stale_nodes}\nKomponenten neu installiert: {components}\nDauer: {duration}\n{warnings_block}\nDer Knoten ist nun vollständig einsatzbereit.",
|
||||||
"label": "Host-Wiederherstellung abgeschlossen"
|
"label": "Host-Wiederherstellung abgeschlossen"
|
||||||
},
|
},
|
||||||
"system_problem": {
|
"system_problem": {
|
||||||
@@ -6910,8 +6910,7 @@
|
|||||||
"warning": "Warnung",
|
"warning": "Warnung",
|
||||||
"info": "Informationen",
|
"info": "Informationen",
|
||||||
"ok": "Gelöst",
|
"ok": "Gelöst",
|
||||||
"default": "Hinweis",
|
"default": "Hinweis"
|
||||||
"observation": "Nicht mehr gemeldet"
|
|
||||||
},
|
},
|
||||||
"groups": {
|
"groups": {
|
||||||
"vm_ct": "Virtuelle Maschine / Container",
|
"vm_ct": "Virtuelle Maschine / Container",
|
||||||
@@ -6992,8 +6991,8 @@
|
|||||||
"temperature": {
|
"temperature": {
|
||||||
"sampleSpan": "Die hohen Messwerte erstrecken sich über {duration}."
|
"sampleSpan": "Die hohen Messwerte erstrecken sich über {duration}."
|
||||||
},
|
},
|
||||||
"healthRecovery": {"title": "{hostname}: Behoben - {category}{entity_suffix}", "body": "Eine aktuelle Zustandsprüfung hat für {category} wieder einen normalen Zustand festgestellt.\nVorherige Beobachtung: {reason}\nVorheriger Schweregrad: {original_severity}\nZeit seit der ersten Beobachtung: {duration}", "status": "Behoben"},
|
|
||||||
"backup": {
|
"backup": {
|
||||||
|
"unconfirmedTitle": "{hostname}: Backup-Ergebnis nicht bestätigt",
|
||||||
"confirmedTitle": "{hostname}: Backup abgeschlossen",
|
"confirmedTitle": "{hostname}: Backup abgeschlossen",
|
||||||
"confirmedBody": "Backup erfolgreich abgeschlossen.",
|
"confirmedBody": "Backup erfolgreich abgeschlossen.",
|
||||||
"errorTitle": "{hostname}: Backup-Fehler gemeldet",
|
"errorTitle": "{hostname}: Backup-Fehler gemeldet",
|
||||||
|
|||||||
File diff suppressed because one or more lines are too long
@@ -6266,8 +6266,8 @@
|
|||||||
"label": "Nuevo problema de salud"
|
"label": "Nuevo problema de salud"
|
||||||
},
|
},
|
||||||
"error_resolved": {
|
"error_resolved": {
|
||||||
"title": "{hostname}: Ya no se informa de {category}{entity_suffix}",
|
"title": "{hostname}: Resuelto - {category}{entity_suffix}",
|
||||||
"body": "El problema de {category} ya no figura entre las incidencias de salud activas.\n{reason}\n🚦 Gravedad anterior: {original_severity}\n⏱️ Tiempo desde la primera observación: {duration}",
|
"body": "El problema {category} se ha resuelto.\n{reason}\n🚦 Gravedad anterior: {original_severity}\n⏱️ Duración: {duration}",
|
||||||
"label": "Notificación de recuperación"
|
"label": "Notificación de recuperación"
|
||||||
},
|
},
|
||||||
"error_escalated": {
|
"error_escalated": {
|
||||||
@@ -6396,8 +6396,8 @@
|
|||||||
"label": "Backup iniciado"
|
"label": "Backup iniciado"
|
||||||
},
|
},
|
||||||
"backup_complete": {
|
"backup_complete": {
|
||||||
"title": "{hostname}: resultado del backup sin confirmar",
|
"title": "{hostname} → {storage}: Backup completado — {vmname} ({vmid})",
|
||||||
"body": "Esta notificación no permite confirmar el resultado del backup.",
|
"body": "El backup de {vmname} (ID: {vmid}) se ha completado correctamente en {storage}.\nTamaño: {size}",
|
||||||
"label": "Backup completado"
|
"label": "Backup completado"
|
||||||
},
|
},
|
||||||
"backup_warning": {
|
"backup_warning": {
|
||||||
@@ -6556,8 +6556,8 @@
|
|||||||
"label": "Reinicio del sistema"
|
"label": "Reinicio del sistema"
|
||||||
},
|
},
|
||||||
"system_restore_completed": {
|
"system_restore_completed": {
|
||||||
"title": "{hostname}: restauración del host finalizada",
|
"title": "{hostname}: Restauración del host finalizada",
|
||||||
"body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nConfiguraciones de guests aplicadas: {guests}\nDirectorios auxiliares de montajes bind: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}",
|
"body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nGuests aplicados: {guests}\nStubs de bind mount: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}\nEl nodo está listo para usarse.",
|
||||||
"label": "Restauración del host completada"
|
"label": "Restauración del host completada"
|
||||||
},
|
},
|
||||||
"system_problem": {
|
"system_problem": {
|
||||||
@@ -6910,8 +6910,7 @@
|
|||||||
"warning": "Advertencia",
|
"warning": "Advertencia",
|
||||||
"info": "Información",
|
"info": "Información",
|
||||||
"ok": "Resuelto",
|
"ok": "Resuelto",
|
||||||
"default": "Aviso",
|
"default": "Aviso"
|
||||||
"observation": "Ya no se informa"
|
|
||||||
},
|
},
|
||||||
"groups": {
|
"groups": {
|
||||||
"vm_ct": "Máquina virtual / Contenedor",
|
"vm_ct": "Máquina virtual / Contenedor",
|
||||||
@@ -6992,11 +6991,11 @@
|
|||||||
"temperature": {
|
"temperature": {
|
||||||
"sampleSpan": "Lecturas altas registradas a lo largo de {duration}."
|
"sampleSpan": "Lecturas altas registradas a lo largo de {duration}."
|
||||||
},
|
},
|
||||||
"healthRecovery": {"title": "{hostname}: Resuelto - {category}{entity_suffix}", "body": "Una comprobación reciente confirma que la condición de {category} volvió a la normalidad.\nObservación anterior: {reason}\nGravedad anterior: {original_severity}\nTiempo desde la primera observación: {duration}", "status": "Resuelto"},
|
|
||||||
"backup": {
|
"backup": {
|
||||||
"confirmedTitle": "{hostname}: backup completado",
|
"unconfirmedTitle": "{hostname}: Resultado del backup sin confirmar",
|
||||||
|
"confirmedTitle": "{hostname}: Backup completado",
|
||||||
"confirmedBody": "Backup completado correctamente.",
|
"confirmedBody": "Backup completado correctamente.",
|
||||||
"errorTitle": "{hostname}: error notificado en el backup",
|
"errorTitle": "{hostname}: Backup fallido",
|
||||||
"errorBody": "El informe del backup contiene un error.",
|
"errorBody": "El informe del backup contiene un error.",
|
||||||
"unconfirmedBody": "El resultado del backup no está confirmado.",
|
"unconfirmedBody": "El resultado del backup no está confirmado.",
|
||||||
"warningTitle": "{hostname}: Backup completado con advertencias",
|
"warningTitle": "{hostname}: Backup completado con advertencias",
|
||||||
|
|||||||
@@ -6266,8 +6266,8 @@
|
|||||||
"label": "Nouveau problème de santé"
|
"label": "Nouveau problème de santé"
|
||||||
},
|
},
|
||||||
"error_resolved": {
|
"error_resolved": {
|
||||||
"title": "{hostname} : Plus signalé – {category}{entity_suffix}",
|
"title": "{hostname} : Résolu - {category}{entity_suffix}",
|
||||||
"body": "Le problème {category} ne figure plus parmi les alertes de santé actives.\n{reason}\n🚦 Gravité précédente : {original_severity}\n⏱️ Temps depuis la première observation : {duration}",
|
"body": "Le problème {category} a été résolu.\n{reason}\n🚦 Gravité précédente : {original_severity}\n⏱️ Durée : {duration}",
|
||||||
"label": "Notification de récupération"
|
"label": "Notification de récupération"
|
||||||
},
|
},
|
||||||
"error_escalated": {
|
"error_escalated": {
|
||||||
@@ -6396,8 +6396,8 @@
|
|||||||
"label": "Sauvegarde démarrée"
|
"label": "Sauvegarde démarrée"
|
||||||
},
|
},
|
||||||
"backup_complete": {
|
"backup_complete": {
|
||||||
"title": "{hostname} : résultat de la sauvegarde non confirmé",
|
"title": "{hostname} → {storage} : Sauvegarde terminée — {vmname} ({vmid})",
|
||||||
"body": "Cette notification ne permet pas de confirmer le résultat de la sauvegarde.",
|
"body": "La sauvegarde de {vmname} (ID : {vmid}) s'est terminée avec succès le {storage}.\nTaille : {size}",
|
||||||
"label": "Sauvegarde terminée"
|
"label": "Sauvegarde terminée"
|
||||||
},
|
},
|
||||||
"backup_warning": {
|
"backup_warning": {
|
||||||
@@ -6557,7 +6557,7 @@
|
|||||||
},
|
},
|
||||||
"system_restore_completed": {
|
"system_restore_completed": {
|
||||||
"title": "{hostname} : restauration de l'hôte terminée",
|
"title": "{hostname} : restauration de l'hôte terminée",
|
||||||
"body": "Tâches après restauration terminées en arrière-plan.\n\nInvités appliqués : {guests}\nRépertoires de support des montages bind : {stubs}\nRépertoires de nœuds obsolètes supprimés : {stale_nodes}\nComposants réinstallés : {components}\nDurée : {duration}\n{warnings_block}",
|
"body": "Tâches post-restauration effectuées en arrière-plan.\n\nInvités postulés : {guests}\nTalons de montage liés : {stubs}\nRépertoires de nœuds obsolètes supprimés : {stale_nodes}\nComposants réinstallés : {components}\nDurée : {duration}\n{warnings_block}\nLe nœud est maintenant entièrement prêt à être utilisé.",
|
||||||
"label": "Restauration de l'hôte terminée"
|
"label": "Restauration de l'hôte terminée"
|
||||||
},
|
},
|
||||||
"system_problem": {
|
"system_problem": {
|
||||||
@@ -6910,8 +6910,7 @@
|
|||||||
"warning": "Avertissement",
|
"warning": "Avertissement",
|
||||||
"info": "Informations",
|
"info": "Informations",
|
||||||
"ok": "Résolu",
|
"ok": "Résolu",
|
||||||
"default": "Avis",
|
"default": "Avis"
|
||||||
"observation": "Plus signalé"
|
|
||||||
},
|
},
|
||||||
"groups": {
|
"groups": {
|
||||||
"vm_ct": "Machine Virtuelle / Conteneur",
|
"vm_ct": "Machine Virtuelle / Conteneur",
|
||||||
@@ -6992,8 +6991,8 @@
|
|||||||
"temperature": {
|
"temperature": {
|
||||||
"sampleSpan": "Les relevés élevés s'étendent sur {duration}."
|
"sampleSpan": "Les relevés élevés s'étendent sur {duration}."
|
||||||
},
|
},
|
||||||
"healthRecovery": {"title": "{hostname} : Résolu - {category}{entity_suffix}", "body": "Un contrôle récent confirme le retour à la normale de la condition {category}.\nObservation précédente : {reason}\nGravité précédente : {original_severity}\nTemps depuis la première observation : {duration}", "status": "Résolu"},
|
|
||||||
"backup": {
|
"backup": {
|
||||||
|
"unconfirmedTitle": "{hostname} : résultat de la sauvegarde non confirmé",
|
||||||
"confirmedTitle": "{hostname} : sauvegarde terminée",
|
"confirmedTitle": "{hostname} : sauvegarde terminée",
|
||||||
"confirmedBody": "Sauvegarde terminée avec succès.",
|
"confirmedBody": "Sauvegarde terminée avec succès.",
|
||||||
"errorTitle": "{hostname} : erreur signalée lors de la sauvegarde",
|
"errorTitle": "{hostname} : erreur signalée lors de la sauvegarde",
|
||||||
|
|||||||
@@ -6266,8 +6266,8 @@
|
|||||||
"label": "Nuovo problema sanitario"
|
"label": "Nuovo problema sanitario"
|
||||||
},
|
},
|
||||||
"error_resolved": {
|
"error_resolved": {
|
||||||
"title": "{hostname}: segnalazione non più attiva - {category}{entity_suffix}",
|
"title": "{hostname}: risolto - {category}{entity_suffix}",
|
||||||
"body": "Il problema relativo a {category} non figura più tra le segnalazioni di salute attive.\n{reason}\n🚦 Gravità precedente: {original_severity}\n⏱️ Tempo dalla prima segnalazione: {duration}",
|
"body": "Il problema {category} è stato risolto.\n{reason}\n🚦 Gravità precedente: {original_severity}\n⏱️ Durata: {duration}",
|
||||||
"label": "Notifica di recupero"
|
"label": "Notifica di recupero"
|
||||||
},
|
},
|
||||||
"error_escalated": {
|
"error_escalated": {
|
||||||
@@ -6396,8 +6396,8 @@
|
|||||||
"label": "Backup avviato"
|
"label": "Backup avviato"
|
||||||
},
|
},
|
||||||
"backup_complete": {
|
"backup_complete": {
|
||||||
"title": "{hostname}: esito del backup non confermato",
|
"title": "{hostname} → {storage}: Backup completato — {vmname} ({vmid})",
|
||||||
"body": "Questa notifica non consente di confermare l’esito del backup.",
|
"body": "Backup di {vmname} (ID: {vmid}) completato con successo su {storage}.\nTaglia: {size}",
|
||||||
"label": "Backup completato"
|
"label": "Backup completato"
|
||||||
},
|
},
|
||||||
"backup_warning": {
|
"backup_warning": {
|
||||||
@@ -6557,7 +6557,7 @@
|
|||||||
},
|
},
|
||||||
"system_restore_completed": {
|
"system_restore_completed": {
|
||||||
"title": "{hostname}: ripristino dell'host terminato",
|
"title": "{hostname}: ripristino dell'host terminato",
|
||||||
"body": "Attività post-ripristino completate in background.\n\nConfigurazioni guest copiate: {guests}\nDirectory di supporto per montaggi bind create: {stubs}\nDirectory obsolete dei nodi rimosse: {stale_nodes}\nComponenti reinstallati: {components}\nDurata: {duration}\n{warnings_block}",
|
"body": "Attività post-ripristino completate in background.\n\nGli ospiti hanno presentato domanda: {guests}\nStub con montaggio tramite collegamento: {stubs}\nDirectory dei nodi obsolete rimosse: {stale_nodes}\nComponenti reinstallati: {components}\nDurata: {duration}\n{warnings_block}\nIl nodo è ora completamente pronto per l'uso.",
|
||||||
"label": "Ripristino dell'host completato"
|
"label": "Ripristino dell'host completato"
|
||||||
},
|
},
|
||||||
"system_problem": {
|
"system_problem": {
|
||||||
@@ -6910,8 +6910,7 @@
|
|||||||
"warning": "Avvertimento",
|
"warning": "Avvertimento",
|
||||||
"info": "Informazioni",
|
"info": "Informazioni",
|
||||||
"ok": "Risolto",
|
"ok": "Risolto",
|
||||||
"default": "Avviso",
|
"default": "Avviso"
|
||||||
"observation": "Non più segnalato"
|
|
||||||
},
|
},
|
||||||
"groups": {
|
"groups": {
|
||||||
"vm_ct": "Macchina virtuale/Contenitore",
|
"vm_ct": "Macchina virtuale/Contenitore",
|
||||||
@@ -6992,8 +6991,8 @@
|
|||||||
"temperature": {
|
"temperature": {
|
||||||
"sampleSpan": "Intervallo dei campioni sopra soglia: {duration}."
|
"sampleSpan": "Intervallo dei campioni sopra soglia: {duration}."
|
||||||
},
|
},
|
||||||
"healthRecovery": {"title": "{hostname}: Risolto - {category}{entity_suffix}", "body": "Un controllo recente conferma che la condizione {category} è tornata nella norma.\nOsservazione precedente: {reason}\nGravità precedente: {original_severity}\nTempo dalla prima osservazione: {duration}", "status": "Risolto"},
|
|
||||||
"backup": {
|
"backup": {
|
||||||
|
"unconfirmedTitle": "{hostname}: esito del backup non confermato",
|
||||||
"confirmedTitle": "{hostname}: backup completato",
|
"confirmedTitle": "{hostname}: backup completato",
|
||||||
"confirmedBody": "Backup completato correttamente.",
|
"confirmedBody": "Backup completato correttamente.",
|
||||||
"errorTitle": "{hostname}: errore segnalato nel backup",
|
"errorTitle": "{hostname}: errore segnalato nel backup",
|
||||||
|
|||||||
@@ -6266,8 +6266,8 @@
|
|||||||
"label": "Novo problema de saúde"
|
"label": "Novo problema de saúde"
|
||||||
},
|
},
|
||||||
"error_resolved": {
|
"error_resolved": {
|
||||||
"title": "{hostname}: Já não comunicado – {category}{entity_suffix}",
|
"title": "{hostname}: Resolvido - {category}{entity_suffix}",
|
||||||
"body": "O problema de {category} já não consta dos registos de saúde ativos.\n{reason}\n🚦 Gravidade anterior: {original_severity}\n⏱️ Tempo desde a primeira observação: {duration}",
|
"body": "O problema {category} foi resolvido.\n{reason}\n🚦 Gravidade anterior: {original_severity}\n⏱️ Duração: {duration}",
|
||||||
"label": "Notificação de recuperação"
|
"label": "Notificação de recuperação"
|
||||||
},
|
},
|
||||||
"error_escalated": {
|
"error_escalated": {
|
||||||
@@ -6396,8 +6396,8 @@
|
|||||||
"label": "Backup iniciado"
|
"label": "Backup iniciado"
|
||||||
},
|
},
|
||||||
"backup_complete": {
|
"backup_complete": {
|
||||||
"title": "{hostname}: resultado do backup não confirmado",
|
"title": "{hostname} → {storage}: Backup concluído — {vmname} ({vmid})",
|
||||||
"body": "Esta notificação não permite confirmar o resultado do backup.",
|
"body": "Backup de {vmname} (ID: {vmid}) concluído com sucesso em {storage}.\nTamanho: {size}",
|
||||||
"label": "Backup concluído"
|
"label": "Backup concluído"
|
||||||
},
|
},
|
||||||
"backup_warning": {
|
"backup_warning": {
|
||||||
@@ -6557,7 +6557,7 @@
|
|||||||
},
|
},
|
||||||
"system_restore_completed": {
|
"system_restore_completed": {
|
||||||
"title": "{hostname}: restauração do host concluída",
|
"title": "{hostname}: restauração do host concluída",
|
||||||
"body": "Tarefas pós-restauro concluídas em segundo plano.\n\nConvidados aplicados: {guests}\nDiretórios auxiliares de montagens bind: {stubs}\nDiretórios de nós obsoletos removidos: {stale_nodes}\nComponentes reinstalados: {components}\nDuração: {duration}\n{warnings_block}",
|
"body": "Tarefas pós-restauração concluídas em segundo plano.\n\nConvidados inscritos: {guests}\nStubs de montagem de ligação: {stubs}\nDiretórios de nó obsoletos removidos: {stale_nodes}\nComponentes reinstalados: {components}\nDuração: {duration}\n{warnings_block}\nO nó agora está totalmente pronto para uso.",
|
||||||
"label": "Restauração do host concluída"
|
"label": "Restauração do host concluída"
|
||||||
},
|
},
|
||||||
"system_problem": {
|
"system_problem": {
|
||||||
@@ -6910,8 +6910,7 @@
|
|||||||
"warning": "Aviso",
|
"warning": "Aviso",
|
||||||
"info": "Informação",
|
"info": "Informação",
|
||||||
"ok": "Resolvido",
|
"ok": "Resolvido",
|
||||||
"default": "Aviso",
|
"default": "Aviso"
|
||||||
"observation": "Já não comunicado"
|
|
||||||
},
|
},
|
||||||
"groups": {
|
"groups": {
|
||||||
"vm_ct": "Máquina Virtual/Contêiner",
|
"vm_ct": "Máquina Virtual/Contêiner",
|
||||||
@@ -6992,8 +6991,8 @@
|
|||||||
"temperature": {
|
"temperature": {
|
||||||
"sampleSpan": "As amostras elevadas abrangem {duration}."
|
"sampleSpan": "As amostras elevadas abrangem {duration}."
|
||||||
},
|
},
|
||||||
"healthRecovery": {"title": "{hostname}: Resolvido - {category}{entity_suffix}", "body": "Uma verificação recente confirma que a condição de {category} voltou ao normal.\nObservação anterior: {reason}\nGravidade anterior: {original_severity}\nTempo desde a primeira observação: {duration}", "status": "Resolvido"},
|
|
||||||
"backup": {
|
"backup": {
|
||||||
|
"unconfirmedTitle": "{hostname}: resultado do backup não confirmado",
|
||||||
"confirmedTitle": "{hostname}: backup concluído",
|
"confirmedTitle": "{hostname}: backup concluído",
|
||||||
"confirmedBody": "Backup concluído com sucesso.",
|
"confirmedBody": "Backup concluído com sucesso.",
|
||||||
"errorTitle": "{hostname}: erro comunicado no backup",
|
"errorTitle": "{hostname}: erro comunicado no backup",
|
||||||
|
|||||||
@@ -6266,8 +6266,8 @@
|
|||||||
"label": "Nytt hälsoproblem"
|
"label": "Nytt hälsoproblem"
|
||||||
},
|
},
|
||||||
"error_resolved": {
|
"error_resolved": {
|
||||||
"title": "{hostname}: Rapporteras inte längre – {category}{entity_suffix}",
|
"title": "{hostname}: Löst - {category}{entity_suffix}",
|
||||||
"body": "Problemet i kategorin {category} finns inte längre bland aktiva hälsoposter.\n{reason}\n🚦 Tidigare allvarlighetsgrad: {original_severity}\n⏱️ Tid sedan första observationen: {duration}",
|
"body": "{category}-problemet har lösts.\n{reason}\n🚦 Tidigare svårighetsgrad: {original_severity}\n⏱️ Varaktighet: {duration}",
|
||||||
"label": "Återställningsmeddelande"
|
"label": "Återställningsmeddelande"
|
||||||
},
|
},
|
||||||
"error_escalated": {
|
"error_escalated": {
|
||||||
@@ -6396,8 +6396,8 @@
|
|||||||
"label": "Säkerhetskopiering startade"
|
"label": "Säkerhetskopiering startade"
|
||||||
},
|
},
|
||||||
"backup_complete": {
|
"backup_complete": {
|
||||||
"title": "{hostname}: säkerhetskopians resultat obekräftat",
|
"title": "{hostname} → {storage}: Säkerhetskopiering klar — {vmname} ({vmid})",
|
||||||
"body": "Det går inte att bekräfta säkerhetskopians resultat utifrån denna avisering.",
|
"body": "Säkerhetskopiering av {vmname} (ID: {vmid}) slutfördes framgångsrikt på {storage}.\nStorlek: {size}",
|
||||||
"label": "Säkerhetskopieringen är klar"
|
"label": "Säkerhetskopieringen är klar"
|
||||||
},
|
},
|
||||||
"backup_warning": {
|
"backup_warning": {
|
||||||
@@ -6557,7 +6557,7 @@
|
|||||||
},
|
},
|
||||||
"system_restore_completed": {
|
"system_restore_completed": {
|
||||||
"title": "{hostname}: Värdåterställning avslutad",
|
"title": "{hostname}: Värdåterställning avslutad",
|
||||||
"body": "Åtgärder efter återställning slutfördes i bakgrunden.\n\nGäster tillämpade: {guests}\nHjälpkataloger för bind-monteringar: {stubs}\nFöråldrade nodkataloger borttagna: {stale_nodes}\nKomponenter ominstallerade: {components}\nVaraktighet: {duration}\n{warnings_block}",
|
"body": "Uppgifter efter återställning slutförda i bakgrunden.\n\nGäster ansökte: {guests}\nBind-monterade stubbar: {stubs}\nInaktuella nodkataloger har tagits bort: {stale_nodes}\nKomponenter installerade om: {components}\nVaraktighet: {duration}\n{warnings_block}\nNoden är nu helt redo att användas.",
|
||||||
"label": "Värdåterställning slutförd"
|
"label": "Värdåterställning slutförd"
|
||||||
},
|
},
|
||||||
"system_problem": {
|
"system_problem": {
|
||||||
@@ -6910,8 +6910,7 @@
|
|||||||
"warning": "Varning",
|
"warning": "Varning",
|
||||||
"info": "Information",
|
"info": "Information",
|
||||||
"ok": "Löst",
|
"ok": "Löst",
|
||||||
"default": "Observera",
|
"default": "Observera"
|
||||||
"observation": "Rapporteras inte längre"
|
|
||||||
},
|
},
|
||||||
"groups": {
|
"groups": {
|
||||||
"vm_ct": "Virtuell maskin / behållare",
|
"vm_ct": "Virtuell maskin / behållare",
|
||||||
@@ -6992,8 +6991,8 @@
|
|||||||
"temperature": {
|
"temperature": {
|
||||||
"sampleSpan": "De höga mätvärdena sträcker sig över {duration}."
|
"sampleSpan": "De höga mätvärdena sträcker sig över {duration}."
|
||||||
},
|
},
|
||||||
"healthRecovery": {"title": "{hostname}: Åtgärdat - {category}{entity_suffix}", "body": "En aktuell hälsokontroll bekräftar att tillståndet för {category} återgått till det normala.\nTidigare observation: {reason}\nTidigare allvarlighetsgrad: {original_severity}\nTid sedan första observationen: {duration}", "status": "Åtgärdat"},
|
|
||||||
"backup": {
|
"backup": {
|
||||||
|
"unconfirmedTitle": "{hostname}: säkerhetskopians resultat obekräftat",
|
||||||
"confirmedTitle": "{hostname}: säkerhetskopiering klar",
|
"confirmedTitle": "{hostname}: säkerhetskopiering klar",
|
||||||
"confirmedBody": "Säkerhetskopieringen slutfördes utan fel.",
|
"confirmedBody": "Säkerhetskopieringen slutfördes utan fel.",
|
||||||
"errorTitle": "{hostname}: fel rapporterat vid säkerhetskopiering",
|
"errorTitle": "{hostname}: fel rapporterat vid säkerhetskopiering",
|
||||||
|
|||||||
@@ -131,7 +131,6 @@ cp "$SCRIPT_DIR/auth_manager.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️
|
|||||||
cp "$SCRIPT_DIR/jwt_middleware.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ jwt_middleware.py not found"
|
cp "$SCRIPT_DIR/jwt_middleware.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ jwt_middleware.py not found"
|
||||||
cp "$SCRIPT_DIR/health_monitor.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_monitor.py not found"
|
cp "$SCRIPT_DIR/health_monitor.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_monitor.py not found"
|
||||||
cp "$SCRIPT_DIR/health_persistence.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_persistence.py not found"
|
cp "$SCRIPT_DIR/health_persistence.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_persistence.py not found"
|
||||||
cp "$SCRIPT_DIR/health_recovery.py" "$APP_DIR/usr/bin/"
|
|
||||||
cp "$SCRIPT_DIR/flask_health_routes.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ flask_health_routes.py not found"
|
cp "$SCRIPT_DIR/flask_health_routes.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ flask_health_routes.py not found"
|
||||||
cp "$SCRIPT_DIR/flask_proxmenux_routes.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ flask_proxmenux_routes.py not found"
|
cp "$SCRIPT_DIR/flask_proxmenux_routes.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ flask_proxmenux_routes.py not found"
|
||||||
cp "$SCRIPT_DIR/post_install_versions.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ post_install_versions.py not found"
|
cp "$SCRIPT_DIR/post_install_versions.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ post_install_versions.py not found"
|
||||||
|
|||||||
@@ -1427,8 +1427,6 @@ class HealthMonitor:
|
|||||||
'details': f'Sustained for {actual_duration}s above {self.CPU_CRITICAL}%.',
|
'details': f'Sustained for {actual_duration}s above {self.CPU_CRITICAL}%.',
|
||||||
'cpu_percent': cpu_percent,
|
'cpu_percent': cpu_percent,
|
||||||
'duration': actual_duration,
|
'duration': actual_duration,
|
||||||
'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
|
|
||||||
'recovery': self.CPU_RECOVERY},
|
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
elif len(warning_samples) >= WARNING_MIN_SAMPLES and len(recovery_samples) < RECOVERY_MIN_SAMPLES:
|
elif len(warning_samples) >= WARNING_MIN_SAMPLES and len(recovery_samples) < RECOVERY_MIN_SAMPLES:
|
||||||
@@ -1448,32 +1446,13 @@ class HealthMonitor:
|
|||||||
'details': f'Sustained for {actual_duration}s above {self.CPU_WARNING}%.',
|
'details': f'Sustained for {actual_duration}s above {self.CPU_WARNING}%.',
|
||||||
'cpu_percent': cpu_percent,
|
'cpu_percent': cpu_percent,
|
||||||
'duration': actual_duration,
|
'duration': actual_duration,
|
||||||
'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
|
|
||||||
'recovery': self.CPU_RECOVERY},
|
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
status = 'OK'
|
status = 'OK'
|
||||||
reason = None
|
reason = None
|
||||||
# CPU is normal - auto-resolve any existing CPU errors
|
# CPU is normal - auto-resolve any existing CPU errors
|
||||||
evidence = None
|
health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal')
|
||||||
# Presentation proof is stricter than operational hysteresis:
|
|
||||||
# a supported warning can be below the fixed recovery cutoff.
|
|
||||||
from health_recovery import _finite_number
|
|
||||||
criterion = min(self.CPU_WARNING, self.CPU_RECOVERY)
|
|
||||||
normal_samples = [entry for entry in self.state_history[state_key]
|
|
||||||
if _finite_number(entry['value']) and 0 <= entry['value'] < criterion
|
|
||||||
and _finite_number(entry['time'])
|
|
||||||
and 0 <= current_time - entry['time'] <= self.CPU_RECOVERY_DURATION]
|
|
||||||
if (_finite_number(cpu_percent) and 0 <= cpu_percent < criterion
|
|
||||||
and len(normal_samples) >= RECOVERY_MIN_SAMPLES):
|
|
||||||
evidence = {'check': 'cpu_usage', 'checked_at': current_time,
|
|
||||||
'value': cpu_percent, 'normal_samples': len(normal_samples),
|
|
||||||
'max_sample': max(entry['value'] for entry in normal_samples),
|
|
||||||
'policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL,
|
|
||||||
'recovery': self.CPU_RECOVERY}}
|
|
||||||
health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal',
|
|
||||||
check_evidence=evidence)
|
|
||||||
|
|
||||||
temp_status = self._check_cpu_temperature()
|
temp_status = self._check_cpu_temperature()
|
||||||
|
|
||||||
@@ -3921,7 +3900,6 @@ class HealthMonitor:
|
|||||||
|
|
||||||
failed_services = []
|
failed_services = []
|
||||||
service_details = {}
|
service_details = {}
|
||||||
active_evidence = {}
|
|
||||||
|
|
||||||
for service in services_to_check:
|
for service in services_to_check:
|
||||||
try:
|
try:
|
||||||
@@ -3936,10 +3914,6 @@ class HealthMonitor:
|
|||||||
if result.returncode != 0 or status != 'active':
|
if result.returncode != 0 or status != 'active':
|
||||||
failed_services.append(service)
|
failed_services.append(service)
|
||||||
service_details[service] = status or 'inactive'
|
service_details[service] = status or 'inactive'
|
||||||
else:
|
|
||||||
active_evidence[service] = {'check': f'pve_service_{service}',
|
|
||||||
'checked_at': time.time(), 'service': service, 'state': status,
|
|
||||||
'returncode': result.returncode}
|
|
||||||
except Exception:
|
except Exception:
|
||||||
failed_services.append(service)
|
failed_services.append(service)
|
||||||
service_details[service] = 'error'
|
service_details[service] = 'error'
|
||||||
@@ -3954,7 +3928,7 @@ class HealthMonitor:
|
|||||||
if svc not in failed_services:
|
if svc not in failed_services:
|
||||||
error_key = f'pve_service_{svc}'
|
error_key = f'pve_service_{svc}'
|
||||||
if health_persistence.is_error_active(error_key):
|
if health_persistence.is_error_active(error_key):
|
||||||
health_persistence.clear_error(error_key, check_evidence=active_evidence.get(svc))
|
health_persistence.clear_error(error_key)
|
||||||
|
|
||||||
# Build checks dict with status per service
|
# Build checks dict with status per service
|
||||||
checks = {}
|
checks = {}
|
||||||
|
|||||||
@@ -600,7 +600,7 @@ class HealthPersistence:
|
|||||||
|
|
||||||
cursor.execute('''
|
cursor.execute('''
|
||||||
SELECT id, acknowledged, resolved_at, category, severity, first_seen,
|
SELECT id, acknowledged, resolved_at, category, severity, first_seen,
|
||||||
notification_sent, suppression_hours, acknowledged_at, details
|
notification_sent, suppression_hours, acknowledged_at
|
||||||
FROM errors WHERE error_key = ?
|
FROM errors WHERE error_key = ?
|
||||||
''', (error_key,))
|
''', (error_key,))
|
||||||
existing = cursor.fetchone()
|
existing = cursor.fetchone()
|
||||||
@@ -609,7 +609,7 @@ class HealthPersistence:
|
|||||||
|
|
||||||
if existing:
|
if existing:
|
||||||
(err_id, ack, resolved_at, old_cat, old_severity, first_seen,
|
(err_id, ack, resolved_at, old_cat, old_severity, first_seen,
|
||||||
notif_sent, stored_suppression, acknowledged_at, old_details_json) = existing
|
notif_sent, stored_suppression, acknowledged_at) = existing
|
||||||
|
|
||||||
if ack == 1:
|
if ack == 1:
|
||||||
# SAFETY OVERRIDE: Critical CPU temperature ALWAYS re-triggers
|
# SAFETY OVERRIDE: Critical CPU temperature ALWAYS re-triggers
|
||||||
@@ -680,18 +680,6 @@ class HealthPersistence:
|
|||||||
conn.commit()
|
conn.commit()
|
||||||
return event_info
|
return event_info
|
||||||
|
|
||||||
# Original CPU policy is immutable for this row's incident.
|
|
||||||
# Never upgrade a legacy row from later/current settings.
|
|
||||||
if error_key == 'cpu_usage':
|
|
||||||
try:
|
|
||||||
old_details = json.loads(old_details_json or '{}')
|
|
||||||
except (ValueError, TypeError):
|
|
||||||
old_details = {}
|
|
||||||
details = dict(details) if isinstance(details, dict) else {}
|
|
||||||
details.pop('cpu_policy', None)
|
|
||||||
if isinstance(old_details, dict) and 'cpu_policy' in old_details:
|
|
||||||
details['cpu_policy'] = old_details['cpu_policy']
|
|
||||||
details_json = json.dumps(details)
|
|
||||||
# Not acknowledged - update existing active error
|
# Not acknowledged - update existing active error
|
||||||
cursor.execute('''
|
cursor.execute('''
|
||||||
UPDATE errors
|
UPDATE errors
|
||||||
@@ -756,12 +744,12 @@ class HealthPersistence:
|
|||||||
|
|
||||||
return event_info
|
return event_info
|
||||||
|
|
||||||
def resolve_error(self, error_key: str, reason: str = 'auto-resolved', *, check_evidence=None):
|
def resolve_error(self, error_key: str, reason: str = 'auto-resolved'):
|
||||||
"""Mark an error as resolved"""
|
"""Mark an error as resolved"""
|
||||||
with self._db_lock:
|
with self._db_lock:
|
||||||
return self._resolve_error_impl(error_key, reason, check_evidence=check_evidence)
|
return self._resolve_error_impl(error_key, reason)
|
||||||
|
|
||||||
def _resolve_error_impl(self, error_key, reason, *, check_evidence=None):
|
def _resolve_error_impl(self, error_key, reason):
|
||||||
with self._db_connection() as conn:
|
with self._db_connection() as conn:
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
now = datetime.now().isoformat()
|
now = datetime.now().isoformat()
|
||||||
@@ -788,7 +776,7 @@ class HealthPersistence:
|
|||||||
# was created — otherwise "Storage 'Tuxis' unavailable"
|
# was created — otherwise "Storage 'Tuxis' unavailable"
|
||||||
# comes back as "Resolved - Storage" with no identity.
|
# comes back as "Resolved - Storage" with no identity.
|
||||||
cursor.execute(
|
cursor.execute(
|
||||||
'SELECT details, id, first_seen, resolved_at FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1',
|
'SELECT details FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1',
|
||||||
(error_key,),
|
(error_key,),
|
||||||
)
|
)
|
||||||
row = cursor.fetchone()
|
row = cursor.fetchone()
|
||||||
@@ -798,71 +786,14 @@ class HealthPersistence:
|
|||||||
stored_details = json.loads(row[0])
|
stored_details = json.loads(row[0])
|
||||||
except Exception:
|
except Exception:
|
||||||
stored_details = None
|
stored_details = None
|
||||||
# Legacy/mismatched original policy cannot certify normality.
|
|
||||||
if error_key == 'cpu_usage' and (not isinstance(stored_details, dict)
|
|
||||||
or not isinstance(check_evidence, dict)
|
|
||||||
or check_evidence.get('policy') != stored_details.get('cpu_policy')
|
|
||||||
or not stored_details.get('cpu_policy')):
|
|
||||||
check_evidence = None
|
|
||||||
self._record_event(cursor, 'resolved', error_key, {
|
self._record_event(cursor, 'resolved', error_key, {
|
||||||
'reason': reason,
|
'reason': reason,
|
||||||
# Only explicit current-check callers attach this proof.
|
|
||||||
# Generic resolve/cleanup remains neutral.
|
|
||||||
'check_evidence': check_evidence,
|
|
||||||
'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': row[3]},
|
|
||||||
'entity': self._entity_from_details(stored_details),
|
'entity': self._entity_from_details(stored_details),
|
||||||
'details': stored_details or {},
|
'details': stored_details or {},
|
||||||
})
|
})
|
||||||
|
|
||||||
conn.commit()
|
conn.commit()
|
||||||
|
|
||||||
def get_recovery_evidence(self, error_key: str, first_seen: str):
|
|
||||||
"""Return fresh same-incident native check proof, never absence of errors.
|
|
||||||
|
|
||||||
Host CPU and exact per-service active checks carry provenance. Other
|
|
||||||
checks, generic clears, excluded/deleted records and legacy events stay
|
|
||||||
neutral until they have equivalent per-condition provenance.
|
|
||||||
"""
|
|
||||||
if not isinstance(error_key, str) or not first_seen or not (
|
|
||||||
error_key == 'cpu_usage' or error_key.startswith('pve_service_')):
|
|
||||||
return None
|
|
||||||
try:
|
|
||||||
# One SQLite statement is one consistent row/ack/closure snapshot.
|
|
||||||
# Latest native observation/closure by durable event id: a later
|
|
||||||
# abnormal record supersedes proof even when a resolved row is
|
|
||||||
# reused and wall-clock time moves backward. No-op clears create
|
|
||||||
# no event, so they do not invalidate a genuine closure.
|
|
||||||
with self._db_lock, self._db_connection() as conn:
|
|
||||||
row = conn.execute('''
|
|
||||||
SELECT e.first_seen, e.last_seen, e.resolved_at, e.acknowledged,
|
|
||||||
e.id, v.timestamp, v.data
|
|
||||||
FROM errors e JOIN events v ON v.id = (
|
|
||||||
SELECT id FROM events WHERE error_key = e.error_key
|
|
||||||
AND event_type IN ('resolved', 'cleared', 'new', 'updated', 'escalated')
|
|
||||||
ORDER BY id DESC LIMIT 1
|
|
||||||
) WHERE e.error_key = ?
|
|
||||||
''', (error_key,)).fetchone()
|
|
||||||
if not row or row[0] != first_seen or not row[2] or row[3]:
|
|
||||||
return None
|
|
||||||
event_data = json.loads(row[6])
|
|
||||||
if event_data.get('incident') != {'id': row[4], 'first_seen': row[0], 'resolved_at': row[2]}:
|
|
||||||
return None
|
|
||||||
proof = event_data.get('check_evidence')
|
|
||||||
from health_recovery import valid_check_evidence
|
|
||||||
if not valid_check_evidence(error_key, proof, now=datetime.now().timestamp()):
|
|
||||||
return None
|
|
||||||
if error_key == 'cpu_usage' and proof.get('policy') != event_data.get('details', {}).get('cpu_policy'):
|
|
||||||
return None
|
|
||||||
checked = float(proof['checked_at'])
|
|
||||||
last_seen = datetime.fromisoformat(row[1]).timestamp()
|
|
||||||
resolved = datetime.fromisoformat(row[2]).timestamp()
|
|
||||||
recorded = datetime.fromisoformat(row[5]).timestamp()
|
|
||||||
if not last_seen <= checked <= resolved <= recorded:
|
|
||||||
return None
|
|
||||||
return proof
|
|
||||||
except (ValueError, TypeError, AttributeError, OverflowError):
|
|
||||||
return None
|
|
||||||
|
|
||||||
def is_error_active(self, error_key: str, category: Optional[str] = None) -> bool:
|
def is_error_active(self, error_key: str, category: Optional[str] = None) -> bool:
|
||||||
"""
|
"""
|
||||||
Check if an error is currently active OR suppressed (dismissed but within suppression period).
|
Check if an error is currently active OR suppressed (dismissed but within suppression period).
|
||||||
@@ -928,7 +859,7 @@ class HealthPersistence:
|
|||||||
|
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def clear_error(self, error_key: str, *, check_evidence=None):
|
def clear_error(self, error_key: str):
|
||||||
"""
|
"""
|
||||||
Remove/resolve a specific error immediately.
|
Remove/resolve a specific error immediately.
|
||||||
Used when the condition that caused the error no longer exists
|
Used when the condition that caused the error no longer exists
|
||||||
@@ -950,7 +881,7 @@ class HealthPersistence:
|
|||||||
|
|
||||||
# Check if this error was acknowledged (dismissed)
|
# Check if this error was acknowledged (dismissed)
|
||||||
cursor.execute('''
|
cursor.execute('''
|
||||||
SELECT acknowledged, id, first_seen FROM errors WHERE error_key = ?
|
SELECT acknowledged FROM errors WHERE error_key = ?
|
||||||
''', (error_key,))
|
''', (error_key,))
|
||||||
row = cursor.fetchone()
|
row = cursor.fetchone()
|
||||||
|
|
||||||
@@ -969,10 +900,7 @@ class HealthPersistence:
|
|||||||
''', (now, error_key))
|
''', (now, error_key))
|
||||||
|
|
||||||
if cursor.rowcount > 0:
|
if cursor.rowcount > 0:
|
||||||
self._record_event(cursor, 'cleared', error_key, {
|
self._record_event(cursor, 'cleared', error_key, {'reason': 'condition_resolved'})
|
||||||
'reason': 'condition_resolved', 'check_evidence': check_evidence,
|
|
||||||
'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': now},
|
|
||||||
})
|
|
||||||
|
|
||||||
conn.commit()
|
conn.commit()
|
||||||
|
|
||||||
|
|||||||
@@ -1,51 +0,0 @@
|
|||||||
"""Bounded recovery metadata admission, without importing monitor singletons.
|
|
||||||
|
|
||||||
Native persistence additionally binds the exact incident and closure. Manual
|
|
||||||
notifications remain authenticated caller assertions: shape validation cannot
|
|
||||||
establish that an asserted measurement actually happened.
|
|
||||||
"""
|
|
||||||
import math
|
|
||||||
import time
|
|
||||||
from typing import TypeGuard
|
|
||||||
|
|
||||||
|
|
||||||
def _finite_number(value) -> TypeGuard[int | float]:
|
|
||||||
try:
|
|
||||||
return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value)
|
|
||||||
except (OverflowError, ValueError, TypeError):
|
|
||||||
return False
|
|
||||||
|
|
||||||
|
|
||||||
def valid_check_evidence(error_key, proof, *, now=None):
|
|
||||||
"""Validate a supported measurement contract, not its external authenticity."""
|
|
||||||
if not isinstance(proof, dict) or proof.get('check') != error_key:
|
|
||||||
return False
|
|
||||||
checked = proof.get('checked_at')
|
|
||||||
now = time.time() if now is None else now
|
|
||||||
if not _finite_number(checked) or not _finite_number(now) or not 0 <= now-checked <= 7200:
|
|
||||||
return False
|
|
||||||
if error_key == 'cpu_usage':
|
|
||||||
policy = proof.get('policy')
|
|
||||||
if not isinstance(policy, dict) or set(policy) != {'warning', 'critical', 'recovery'}:
|
|
||||||
return False
|
|
||||||
if not all(_finite_number(v) and 1 <= v <= 100 for v in policy.values()):
|
|
||||||
return False
|
|
||||||
if policy['warning'] > policy['critical']:
|
|
||||||
return False
|
|
||||||
value, maximum, count = proof.get('value'), proof.get('max_sample'), proof.get('normal_samples')
|
|
||||||
return (_finite_number(value) and _finite_number(maximum)
|
|
||||||
and 0 <= value <= maximum < min(policy['warning'], policy['recovery'])
|
|
||||||
and isinstance(count, int) and not isinstance(count, bool) and count >= 10)
|
|
||||||
if isinstance(error_key, str) and error_key.startswith('pve_service_'):
|
|
||||||
service = error_key[len('pve_service_'):]
|
|
||||||
return (bool(service) and proof.get('service') == service
|
|
||||||
and proof.get('state') == 'active'
|
|
||||||
and type(proof.get('returncode')) is int and proof['returncode'] == 0)
|
|
||||||
return False
|
|
||||||
|
|
||||||
|
|
||||||
def presents_recovery(data):
|
|
||||||
"""One presentation predicate shared by template, icon and email badge."""
|
|
||||||
return (isinstance(data, dict) and data.get('recovery_outcome') == 'resolved'
|
|
||||||
and data.get('is_recovery') is True
|
|
||||||
and valid_check_evidence(data.get('error_key'), data.get('check_evidence')))
|
|
||||||
@@ -1037,15 +1037,7 @@ class EmailChannel(NotificationChannel):
|
|||||||
|
|
||||||
# Determine group for section header
|
# Determine group for section header
|
||||||
event_type = data.get('_event_type', '')
|
event_type = data.get('_event_type', '')
|
||||||
if event_type == 'error_resolved':
|
if event_type == 'backup_complete':
|
||||||
from health_recovery import presents_recovery
|
|
||||||
if presents_recovery(data):
|
|
||||||
sev.update(self._SEV_STYLE['OK'])
|
|
||||||
sev['label'] = _runtime_notification_text('healthRecovery.status', data)
|
|
||||||
else:
|
|
||||||
sev.update(self._SEV_DEFAULT)
|
|
||||||
sev['label'] = _runtime_text('email.severity.observation', data)
|
|
||||||
elif event_type == 'backup_complete':
|
|
||||||
outcome = data.get('backup_outcome')
|
outcome = data.get('backup_outcome')
|
||||||
if outcome == 'confirmed':
|
if outcome == 'confirmed':
|
||||||
sev.update(self._SEV_STYLE['OK'])
|
sev.update(self._SEV_STYLE['OK'])
|
||||||
@@ -1064,13 +1056,12 @@ class EmailChannel(NotificationChannel):
|
|||||||
# Scoped inline mail-compatible wrapping for authoritative raw-context
|
# Scoped inline mail-compatible wrapping for authoritative raw-context
|
||||||
# bodies, including restore bodies released from quiet hours.
|
# bodies, including restore bodies released from quiet hours.
|
||||||
backup_email = event_type in {'backup_complete', 'backup_fail'}
|
backup_email = event_type in {'backup_complete', 'backup_fail'}
|
||||||
wrap_body = (event_type in {'temp_high', 'system_restore_completed', 'error_resolved'}
|
wrap_body = (event_type in {'temp_high', 'system_restore_completed'}
|
||||||
or backup_email or data.get('_restore_summary') or data.get('_backup_summary'))
|
or backup_email or data.get('_restore_summary') or data.get('_backup_summary'))
|
||||||
temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else ''
|
temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else ''
|
||||||
temp_table_layout = 'table-layout:fixed;' if wrap_body else ''
|
temp_table_layout = 'table-layout:fixed;' if wrap_body else ''
|
||||||
# Recovery exposes the same literal host context as backup notices.
|
|
||||||
# Keep wrapping event-scoped; unrelated mail remains byte-identical.
|
# Keep wrapping event-scoped; unrelated mail remains byte-identical.
|
||||||
context_email = backup_email or event_type == 'error_resolved'
|
context_email = backup_email
|
||||||
backup_title_wrap = temp_cell_wrap if context_email else ''
|
backup_title_wrap = temp_cell_wrap if context_email else ''
|
||||||
backup_metadata_layout = 'table-layout:fixed;' if context_email else ''
|
backup_metadata_layout = 'table-layout:fixed;' if context_email else ''
|
||||||
section_label = _runtime_text(f'email.groups.{group}', data)
|
section_label = _runtime_text(f'email.groups.{group}', data)
|
||||||
@@ -1106,12 +1097,11 @@ class EmailChannel(NotificationChannel):
|
|||||||
# A metadata-only/manual body may be generic. Keep actionable raw
|
# A metadata-only/manual body may be generic. Keep actionable raw
|
||||||
# context once, without restoring duplicated inventory metadata.
|
# context once, without restoring duplicated inventory metadata.
|
||||||
reason = data.get('reason', '')
|
reason = data.get('reason', '')
|
||||||
if reason and len(reason) <= 80 and reason not in body:
|
if backup_email and reason and len(reason) <= 80 and reason not in body:
|
||||||
detail_rows.append((html_mod.escape(_runtime_text('email.fields.reason', data)),
|
detail_rows.append((html_mod.escape(_runtime_text('email.fields.reason', data)),
|
||||||
html_mod.escape(reason)))
|
html_mod.escape(reason)))
|
||||||
|
|
||||||
if event_type in {'system_restore_completed', 'error_resolved'} or data.get('_restore_summary'):
|
if event_type == 'system_restore_completed' or data.get('_restore_summary'):
|
||||||
# Observation age/disappearance must not become a green OK row.
|
|
||||||
# The endpoint's warnings_block and task counts live in the
|
# The endpoint's warnings_block and task counts live in the
|
||||||
# localized body, not the generic services Event row.
|
# localized body, not the generic services Event row.
|
||||||
detail_rows = [('', html_mod.escape(line if data.get('_quiet_hours_summary') else line.strip()))
|
detail_rows = [('', html_mod.escape(line if data.get('_quiet_hours_summary') else line.strip()))
|
||||||
@@ -1150,7 +1140,7 @@ class EmailChannel(NotificationChannel):
|
|||||||
reason = data.get('reason', '')
|
reason = data.get('reason', '')
|
||||||
reason_html = ''
|
reason_html = ''
|
||||||
if reason and len(reason) > 80 and not (
|
if reason and len(reason) > 80 and not (
|
||||||
(event_type in {'temp_high', 'error_resolved'} or backup_email) and reason in body):
|
(event_type == 'temp_high' or backup_email) and reason in body):
|
||||||
reason_html = f'''
|
reason_html = f'''
|
||||||
<div style="margin:16px 0 0;padding:12px 16px;border:1px solid #d1d5db;border-radius:6px;">
|
<div style="margin:16px 0 0;padding:12px 16px;border:1px solid #d1d5db;border-radius:6px;">
|
||||||
<p style="margin:0 0 4px;font-size:11px;font-weight:600;color:#374151;text-transform:uppercase;letter-spacing:0.05em;">{_runtime_text('email.details', data)}</p>
|
<p style="margin:0 0 4px;font-size:11px;font-weight:600;color:#374151;text-transform:uppercase;letter-spacing:0.05em;">{_runtime_text('email.details', data)}</p>
|
||||||
|
|||||||
@@ -3223,13 +3223,6 @@ class PollingCollector:
|
|||||||
self._last_notified.pop(key, None)
|
self._last_notified.pop(key, None)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Disappearance is not recovery. Only same-incident proof written
|
|
||||||
# by a successful existing native check can certify normality.
|
|
||||||
try:
|
|
||||||
recovery_evidence = health_persistence.get_recovery_evidence(key, first_seen)
|
|
||||||
except Exception:
|
|
||||||
recovery_evidence = None
|
|
||||||
|
|
||||||
# Calculate duration
|
# Calculate duration
|
||||||
duration = ''
|
duration = ''
|
||||||
if first_seen:
|
if first_seen:
|
||||||
@@ -3261,7 +3254,7 @@ class PollingCollector:
|
|||||||
reason_lines = (reason or '').split('\n')
|
reason_lines = (reason or '').split('\n')
|
||||||
reason_summary = reason_lines[0] if reason_lines else ''
|
reason_summary = reason_lines[0] if reason_lines else ''
|
||||||
|
|
||||||
# Keep the earlier device context without asserting recovery.
|
# Try to extract device info for a clean "Device: xxx (recovered)" line
|
||||||
device_line = ''
|
device_line = ''
|
||||||
for line in reason_lines:
|
for line in reason_lines:
|
||||||
if 'Device:' in line or 'Device not currently' in line or '/dev/' in line:
|
if 'Device:' in line or 'Device not currently' in line or '/dev/' in line:
|
||||||
@@ -3274,14 +3267,11 @@ class PollingCollector:
|
|||||||
break
|
break
|
||||||
|
|
||||||
if reason_summary and device_line:
|
if reason_summary and device_line:
|
||||||
clean_reason = f'{reason_summary}\n{device_line} (no longer reported)'
|
clean_reason = f'{reason_summary}\n{device_line} (recovered)'
|
||||||
elif reason_summary:
|
elif reason_summary:
|
||||||
clean_reason = f'{reason_summary} (no longer reported)'
|
clean_reason = f'{reason_summary} (recovered)'
|
||||||
else:
|
else:
|
||||||
clean_reason = 'Condition no longer reported'
|
clean_reason = 'Condition resolved'
|
||||||
|
|
||||||
if recovery_evidence:
|
|
||||||
clean_reason = reason_summary
|
|
||||||
|
|
||||||
# `original_severity` must match what the user actually saw
|
# `original_severity` must match what the user actually saw
|
||||||
# in the most-recent notification for this error, not the
|
# in the most-recent notification for this error, not the
|
||||||
@@ -3309,9 +3299,7 @@ class PollingCollector:
|
|||||||
'original_severity': original_severity,
|
'original_severity': original_severity,
|
||||||
'first_seen': first_seen,
|
'first_seen': first_seen,
|
||||||
'duration': duration_label,
|
'duration': duration_label,
|
||||||
'is_recovery': bool(recovery_evidence),
|
'is_recovery': True,
|
||||||
'recovery_outcome': 'resolved' if recovery_evidence else 'no_longer_reported',
|
|
||||||
'check_evidence': recovery_evidence,
|
|
||||||
}
|
}
|
||||||
# Spread the original details blob so the resolved notification
|
# Spread the original details blob so the resolved notification
|
||||||
# can use the same {storage_name}/{vm_name}/{device} placeholders
|
# can use the same {storage_name}/{vm_name}/{device} placeholders
|
||||||
|
|||||||
@@ -2012,7 +2012,9 @@ class NotificationManager:
|
|||||||
'digest.quietTitle', language, hostname=host, count=len(rows),
|
'digest.quietTitle', language, hostname=host, count=len(rows),
|
||||||
)
|
)
|
||||||
use_icons = self._config.get(f'{ch_name}.rich_format', 'false') == 'true'
|
use_icons = self._config.get(f'{ch_name}.rich_format', 'false') == 'true'
|
||||||
summary_body = self._compose_digest_body(rows, use_icons=use_icons, quiet_release=True)
|
quiet_details = any(row[1] in ('backup_complete', 'backup_fail', 'system_restore_completed')
|
||||||
|
for row in rows)
|
||||||
|
summary_body = self._compose_digest_body(rows, use_icons=use_icons, quiet_release=quiet_details)
|
||||||
|
|
||||||
result: dict = {'success': False, 'error': ''}
|
result: dict = {'success': False, 'error': ''}
|
||||||
try:
|
try:
|
||||||
@@ -2498,7 +2500,7 @@ class NotificationManager:
|
|||||||
runtime_data.setdefault('_notification_language', self._notification_language())
|
runtime_data.setdefault('_notification_language', self._notification_language())
|
||||||
# Match queued dispatch's presentation context for these outcome
|
# Match queued dispatch's presentation context for these outcome
|
||||||
# notices; this does not alter event/severity or direct-send policy.
|
# notices; this does not alter event/severity or direct-send policy.
|
||||||
if event_type in ('backup_complete', 'backup_fail', 'error_resolved', 'system_restore_completed'):
|
if event_type in ('backup_complete', 'backup_fail', 'system_restore_completed'):
|
||||||
runtime_data['_event_type'] = event_type
|
runtime_data['_event_type'] = event_type
|
||||||
runtime_data['_group'] = TEMPLATES[event_type].get('group', 'other')
|
runtime_data['_group'] = TEMPLATES[event_type].get('group', 'other')
|
||||||
|
|
||||||
|
|||||||
@@ -817,10 +817,10 @@ TEMPLATES = {
|
|||||||
# `{entity}` is populated by health_persistence.resolve_error()
|
# `{entity}` is populated by health_persistence.resolve_error()
|
||||||
# (via _entity_from_details) and by PollingCollector's spread of
|
# (via _entity_from_details) and by PollingCollector's spread of
|
||||||
# the original details blob. When absent, _SafeDict elides the
|
# the original details blob. When absent, _SafeDict elides the
|
||||||
# placeholder and the title collapses back to "No longer reported - <cat>"
|
# placeholder and the title collapses back to "Resolved - <cat>"
|
||||||
# without a trailing dash.
|
# without a trailing dash.
|
||||||
'title': '{hostname}: No longer reported - {category}{entity_suffix}',
|
'title': '{hostname}: Resolved - {category}{entity_suffix}',
|
||||||
'body': 'The {category} issue is no longer in active health records.\n{reason}\n\U0001F6A6 Previous severity: {original_severity}\n\u23F1\uFE0F Time since first observation: {duration}',
|
'body': 'The {category} issue has been resolved.\n{reason}\n\U0001F6A6 Previous severity: {original_severity}\n\u23F1\uFE0F Duration: {duration}',
|
||||||
'label': 'Recovery notification',
|
'label': 'Recovery notification',
|
||||||
'group': 'health',
|
'group': 'health',
|
||||||
'default_enabled': True,
|
'default_enabled': True,
|
||||||
@@ -1037,8 +1037,8 @@ TEMPLATES = {
|
|||||||
'default_enabled': False,
|
'default_enabled': False,
|
||||||
},
|
},
|
||||||
'backup_complete': {
|
'backup_complete': {
|
||||||
'title': '{hostname}: Backup outcome unconfirmed',
|
'title': '{hostname} → {storage}: Backup complete — {vmname} ({vmid})',
|
||||||
'body': 'The backup outcome could not be confirmed from this notice.',
|
'body': 'Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}',
|
||||||
'label': 'Backup complete',
|
'label': 'Backup complete',
|
||||||
'group': 'backup',
|
'group': 'backup',
|
||||||
'default_enabled': True,
|
'default_enabled': True,
|
||||||
@@ -1308,7 +1308,8 @@ TEMPLATES = {
|
|||||||
'Stale node dirs removed: {stale_nodes}\n'
|
'Stale node dirs removed: {stale_nodes}\n'
|
||||||
'Components reinstalled: {components}\n'
|
'Components reinstalled: {components}\n'
|
||||||
'Duration: {duration}\n'
|
'Duration: {duration}\n'
|
||||||
'{warnings_block}'
|
'{warnings_block}\n'
|
||||||
|
'The node is now fully ready to use.'
|
||||||
),
|
),
|
||||||
'label': 'Host restore completed',
|
'label': 'Host restore completed',
|
||||||
'group': 'services',
|
'group': 'services',
|
||||||
@@ -1894,6 +1895,10 @@ def render_template(event_type: str, data: Dict[str, Any],
|
|||||||
template[field] = localized
|
template[field] = localized
|
||||||
backup_title_target = ''
|
backup_title_target = ''
|
||||||
if event_type == 'backup_complete':
|
if event_type == 'backup_complete':
|
||||||
|
template['title'] = runtime_message('backup.unconfirmedTitle', language,
|
||||||
|
hostname=data.get('hostname') or _get_hostname()) or (
|
||||||
|
str(data.get('hostname') or _get_hostname()) + ': Backup outcome unconfirmed')
|
||||||
|
template['body'] = runtime_message('backup.unconfirmedBody', language) or 'The backup outcome is not confirmed.'
|
||||||
outcome = data.get('backup_outcome')
|
outcome = data.get('backup_outcome')
|
||||||
if outcome == 'confirmed':
|
if outcome == 'confirmed':
|
||||||
template['title'] = runtime_message('backup.confirmedTitle', language,
|
template['title'] = runtime_message('backup.confirmedTitle', language,
|
||||||
@@ -1961,7 +1966,7 @@ def render_template(event_type: str, data: Dict[str, Any],
|
|||||||
'log_file': '',
|
'log_file': '',
|
||||||
}
|
}
|
||||||
variables.update(data)
|
variables.update(data)
|
||||||
if event_type == 'backup_fail' or (event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'completed_with_warnings', 'failed')):
|
if event_type in ('backup_fail', 'backup_complete'):
|
||||||
# The provider has already substituted raw Display Names. Insert the
|
# The provider has already substituted raw Display Names. Insert the
|
||||||
# resolved title as a value, never reinterpret its literal braces.
|
# resolved title as a value, never reinterpret its literal braces.
|
||||||
variables['_backup_title'] = template['title']
|
variables['_backup_title'] = template['title']
|
||||||
@@ -2070,13 +2075,6 @@ def render_template(event_type: str, data: Dict[str, Any],
|
|||||||
return ''
|
return ''
|
||||||
|
|
||||||
safe_vars = _SafeDict(variables)
|
safe_vars = _SafeDict(variables)
|
||||||
if event_type == 'error_resolved':
|
|
||||||
from health_recovery import presents_recovery
|
|
||||||
if presents_recovery(data):
|
|
||||||
safe_vars['_health_title'] = runtime_message('healthRecovery.title', language, **variables)
|
|
||||||
safe_vars['_health_body'] = runtime_message('healthRecovery.body', language, **variables)
|
|
||||||
template['title'] = '{_health_title}'
|
|
||||||
template['body'] = '{_health_body}'
|
|
||||||
try:
|
try:
|
||||||
title = template['title'].format_map(safe_vars)
|
title = template['title'].format_map(safe_vars)
|
||||||
except (ValueError, IndexError):
|
except (ValueError, IndexError):
|
||||||
@@ -2152,7 +2150,7 @@ def render_template(event_type: str, data: Dict[str, Any],
|
|||||||
key = ('backup.errorBody' if data.get('backup_outcome') == 'failed'
|
key = ('backup.errorBody' if data.get('backup_outcome') == 'failed'
|
||||||
else 'backup.warningBody' if data.get('backup_outcome') == 'completed_with_warnings'
|
else 'backup.warningBody' if data.get('backup_outcome') == 'completed_with_warnings'
|
||||||
else 'backup.unconfirmedBody')
|
else 'backup.unconfirmedBody')
|
||||||
body_text = runtime_message(key, language) + '\n' + body_text
|
body_text = (runtime_message(key, language) or template['body']) + '\n' + body_text
|
||||||
elif event_type == 'system_mail' and pve_message:
|
elif event_type == 'system_mail' and pve_message:
|
||||||
# System mail -- use PVE message directly (mail bounce, cron, smartd)
|
# System mail -- use PVE message directly (mail bounce, cron, smartd)
|
||||||
body_text = pve_message.strip()[:1000]
|
body_text = pve_message.strip()[:1000]
|
||||||
@@ -2181,14 +2179,19 @@ def render_template(event_type: str, data: Dict[str, Any],
|
|||||||
if event_type in ('backup_complete', 'backup_fail') and (
|
if event_type in ('backup_complete', 'backup_fail') and (
|
||||||
event_type == 'backup_fail' or data.get('backup_outcome') == 'failed'):
|
event_type == 'backup_fail' or data.get('backup_outcome') == 'failed'):
|
||||||
source_subject = str(data.get('pve_title') or '').strip()
|
source_subject = str(data.get('pve_title') or '').strip()
|
||||||
|
native_failure = re.fullmatch(
|
||||||
|
r'vzdump backup status \([^\r\n]*\): backup failed(?::\s*(.*))?',
|
||||||
|
source_subject, re.IGNORECASE)
|
||||||
guest_context = (_parse_vzdump_message(str(pve_message or '')) or {}).get('vms')
|
guest_context = (_parse_vzdump_message(str(pve_message or '')) or {}).get('vms')
|
||||||
if guest_context:
|
if native_failure:
|
||||||
# Native single-line job errors live only in the subject. Retain
|
# Before the first guest, too, only the native cause is diagnostic;
|
||||||
# that cause, not the redundant job/host envelope or generic count.
|
# the original host/job envelope is not display-name context.
|
||||||
|
source_subject = (native_failure.group(1) or '').strip()
|
||||||
|
elif guest_context:
|
||||||
cause = re.search(r'\bbackup failed:\s*(.+)', source_subject, re.IGNORECASE)
|
cause = re.search(r'\bbackup failed:\s*(.+)', source_subject, re.IGNORECASE)
|
||||||
source_subject = cause.group(1).strip() if cause else ''
|
source_subject = cause.group(1).strip() if cause else ''
|
||||||
if source_subject.lower() == 'multiple problems':
|
if source_subject.lower() == 'multiple problems':
|
||||||
source_subject = ''
|
source_subject = ''
|
||||||
if source_subject and source_subject not in {line.strip() for line in body_text.splitlines()}:
|
if source_subject and source_subject not in {line.strip() for line in body_text.splitlines()}:
|
||||||
# Reserve the subject-equivalent diagnostic BEFORE the cap. Finding
|
# Reserve the subject-equivalent diagnostic BEFORE the cap. Finding
|
||||||
# it in uncapped logs is not enough: that late line could be omitted.
|
# it in uncapped logs is not enough: that late line could be omitted.
|
||||||
@@ -2374,14 +2377,14 @@ EVENT_EMOJI = {
|
|||||||
'system_startup': '\U0001F680', # rocket (startup)
|
'system_startup': '\U0001F680', # rocket (startup)
|
||||||
'system_shutdown': '\u23FB\uFE0F', # power symbol (Unicode)
|
'system_shutdown': '\u23FB\uFE0F', # power symbol (Unicode)
|
||||||
'system_reboot': '\U0001F504',
|
'system_reboot': '\U0001F504',
|
||||||
'system_restore_completed': '\U0001F4CB', # post-restore task report (boot may have warnings)
|
'system_restore_completed': '✅', # check mark
|
||||||
'system_problem': '\u26A0\uFE0F',
|
'system_problem': '\u26A0\uFE0F',
|
||||||
'kernel_warning': '\u26A0\uFE0F',
|
'kernel_warning': '\u26A0\uFE0F',
|
||||||
'service_fail': '\u274C',
|
'service_fail': '\u274C',
|
||||||
'oom_kill': '\U0001F4A3', # bomb
|
'oom_kill': '\U0001F4A3', # bomb
|
||||||
# Health
|
# Health
|
||||||
'new_error': '\U0001F198', # SOS
|
'new_error': '\U0001F198', # SOS
|
||||||
'error_resolved': '\U0001F4CB', # no longer active in health records, not proven recovery
|
'error_resolved': '\u2705',
|
||||||
'error_escalated': '\U0001F53A', # red triangle up
|
'error_escalated': '\U0001F53A', # red triangle up
|
||||||
'health_degraded': '\u26A0\uFE0F',
|
'health_degraded': '\u26A0\uFE0F',
|
||||||
'health_persistent': '\U0001F4CB', # clipboard
|
'health_persistent': '\U0001F4CB', # clipboard
|
||||||
@@ -2533,10 +2536,6 @@ def enrich_with_emojis(event_type: str, title: str, body: str,
|
|||||||
severity = data.get('severity', 'INFO')
|
severity = data.get('severity', 'INFO')
|
||||||
|
|
||||||
icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '')
|
icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '')
|
||||||
if event_type == 'error_resolved':
|
|
||||||
from health_recovery import presents_recovery
|
|
||||||
if presents_recovery(data):
|
|
||||||
icon = '✅'
|
|
||||||
if event_type == 'backup_complete':
|
if event_type == 'backup_complete':
|
||||||
icon = {
|
icon = {
|
||||||
'confirmed': '💾✅', 'completed_with_warnings': '💾⚠️', 'failed': '💾❌',
|
'confirmed': '💾✅', 'completed_with_warnings': '💾⚠️', 'failed': '💾❌',
|
||||||
|
|||||||
@@ -50,10 +50,6 @@ class RuntimeCatalogTests(unittest.TestCase):
|
|||||||
self.assertIsInstance(templates[event_type][field], str)
|
self.assertIsInstance(templates[event_type][field], str)
|
||||||
self.assertTrue(templates[event_type][field])
|
self.assertTrue(templates[event_type][field])
|
||||||
if field in source:
|
if field in source:
|
||||||
# Upstream Slovak backup title/body still belong to
|
|
||||||
# the pre-outcome schema; runtime falls back to EN.
|
|
||||||
if language == 'sk' and event_type == 'backup_complete' and field != 'label':
|
|
||||||
continue
|
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
_placeholders(templates[event_type][field]),
|
_placeholders(templates[event_type][field]),
|
||||||
_placeholders(source[field]),
|
_placeholders(source[field]),
|
||||||
@@ -74,10 +70,9 @@ class RuntimeCatalogTests(unittest.TestCase):
|
|||||||
en = flatten(self.catalogs["en"])
|
en = flatten(self.catalogs["en"])
|
||||||
pending_slovak = {"backup.confirmedTitle", "backup.confirmedBody",
|
pending_slovak = {"backup.confirmedTitle", "backup.confirmedBody",
|
||||||
"backup.errorTitle", "backup.errorBody", "backup.unconfirmedBody",
|
"backup.errorTitle", "backup.errorBody", "backup.unconfirmedBody",
|
||||||
"channels.email.severity.observation", "channels.email.status.unconfirmed",
|
"backup.unconfirmedTitle", "channels.email.status.unconfirmed",
|
||||||
"backup.warningTitle", "backup.warningBody", "backup.diagnosticsOmitted",
|
"backup.warningTitle", "backup.warningBody", "backup.diagnosticsOmitted",
|
||||||
"channels.email.status.completed_with_warnings",
|
"channels.email.status.completed_with_warnings"}
|
||||||
"healthRecovery.title", "healthRecovery.body", "healthRecovery.status"}
|
|
||||||
for language, catalog in self.catalogs.items():
|
for language, catalog in self.catalogs.items():
|
||||||
translated = flatten(catalog)
|
translated = flatten(catalog)
|
||||||
if language == 'sk':
|
if language == 'sk':
|
||||||
@@ -90,9 +85,6 @@ class RuntimeCatalogTests(unittest.TestCase):
|
|||||||
for key in translated:
|
for key in translated:
|
||||||
self.assertIsInstance(translated[key], str, f"{language}:{key}")
|
self.assertIsInstance(translated[key], str, f"{language}:{key}")
|
||||||
self.assertTrue(translated[key].strip(), f"{language}:{key}")
|
self.assertTrue(translated[key].strip(), f"{language}:{key}")
|
||||||
if language == 'sk' and key in ('templates.backup_complete.title',
|
|
||||||
'templates.backup_complete.body'):
|
|
||||||
continue # exact upstream SK, superseded only at render time
|
|
||||||
self.assertEqual(_placeholders(translated[key]), _placeholders(en[key]), f"{language}:{key}")
|
self.assertEqual(_placeholders(translated[key]), _placeholders(en[key]), f"{language}:{key}")
|
||||||
|
|
||||||
def test_notification_language_ui_keys_exist_in_both_catalogs(self):
|
def test_notification_language_ui_keys_exist_in_both_catalogs(self):
|
||||||
@@ -132,6 +124,11 @@ class RuntimeCatalogTests(unittest.TestCase):
|
|||||||
with mock.patch.object(notification_templates, "_get_hostname", return_value="HOST-ŽILINA"):
|
with mock.patch.object(notification_templates, "_get_hostname", return_value="HOST-ŽILINA"):
|
||||||
for event_type in notification_templates.TEMPLATES:
|
for event_type in notification_templates.TEMPLATES:
|
||||||
event_values = dict(values)
|
event_values = dict(values)
|
||||||
|
if event_type == "backup_complete":
|
||||||
|
# Legacy catalog fields keep their completion meaning;
|
||||||
|
# the new unknown-outcome keys intentionally omit guest data.
|
||||||
|
# Exercise retained metadata on the explicit confirmed path.
|
||||||
|
event_values["backup_outcome"] = "confirmed"
|
||||||
if event_type == "temp_high":
|
if event_type == "temp_high":
|
||||||
# Temperature is a measured numeric contract; arbitrary
|
# Temperature is a measured numeric contract; arbitrary
|
||||||
# DYNAMIC_VALUE is correctly rejected by its fallback.
|
# DYNAMIC_VALUE is correctly rejected by its fallback.
|
||||||
|
|||||||
Reference in New Issue
Block a user