From 08598d2863f8afaa5b6cfc0240c55f952f23a41f Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:48:37 +0200 Subject: [PATCH] fix(notifications): preserve evidenced outcomes in backup and health notices --- .../tests/test_command_descriptions.py | 4 + .../tests/test_notification_corrections.py | 4 +- .../test_notification_maintainer_followup.py | 150 ++++++++++++++++++ .../test_notification_outcome_wording.py | 8 +- .../scripts/tests/test_notification_pve92.py | 4 +- .../test_notification_recovery_evidence.py | 107 +++++++++++++ AppImage/messages/de/common.json | 9 +- AppImage/messages/en/common.json | 2 +- AppImage/messages/es/common.json | 9 +- AppImage/messages/fr/common.json | 9 +- AppImage/messages/it/common.json | 9 +- AppImage/messages/pt/common.json | 9 +- AppImage/messages/sv/common.json | 9 +- AppImage/scripts/health_monitor.py | 6 +- AppImage/scripts/health_persistence.py | 53 ++++++- AppImage/scripts/notification_channels.py | 22 ++- AppImage/scripts/notification_events.py | 26 ++- AppImage/scripts/notification_manager.py | 23 ++- AppImage/scripts/notification_templates.py | 80 ++++++++-- .../tests/test_notification_runtime_i18n.py | 5 +- 20 files changed, 497 insertions(+), 51 deletions(-) create mode 100644 .github/scripts/tests/test_notification_maintainer_followup.py create mode 100644 .github/scripts/tests/test_notification_recovery_evidence.py diff --git a/.github/scripts/tests/test_command_descriptions.py b/.github/scripts/tests/test_command_descriptions.py index be4ad419..16e0e9a2 100644 --- a/.github/scripts/tests/test_command_descriptions.py +++ b/.github/scripts/tests/test_command_descriptions.py @@ -142,6 +142,10 @@ class CommandDescriptionsTests(unittest.TestCase): 'observation', source['channels']['email']['severity']['observation']) local['channels']['email']['status'].setdefault( 'unconfirmed', source['channels']['email']['status']['unconfirmed']) + local['channels']['email']['status'].setdefault( + 'completed_with_warnings', source['channels']['email']['status']['completed_with_warnings']) + for key, value in source['healthRecovery'].items(): + local.setdefault('healthRecovery', {}).setdefault(key, value) path.write_text(json.dumps(temporary, ensure_ascii=False)) # Model steady state after the bot fills these intentional new # messages; keep repository locales and all other leaves intact. diff --git a/.github/scripts/tests/test_notification_corrections.py b/.github/scripts/tests/test_notification_corrections.py index 94ae1469..c9bfaa22 100644 --- a/.github/scripts/tests/test_notification_corrections.py +++ b/.github/scripts/tests/test_notification_corrections.py @@ -217,10 +217,10 @@ class CorrectionTests(unittest.TestCase): self.assertIn('✅ VM web (100)', result['body']) - def test_official_prefixed_warning_is_uncertain_and_retained(self): + def test_official_completed_report_warning_is_distinct_and_retained(self): warning = '100: 2026-09-29 17:00:00 WARN: unable to add notes - permission denied' event = receive(REPORT + '\n' + warning) - self.assertEqual(event.data['backup_outcome'], 'unconfirmed') + self.assertEqual(event.data['backup_outcome'], 'completed_with_warnings') for language in LANGUAGES: result, markup = email(event.event_type, event.data, event.severity, language) self.assertIn(warning, result['body']) diff --git a/.github/scripts/tests/test_notification_maintainer_followup.py b/.github/scripts/tests/test_notification_maintainer_followup.py new file mode 100644 index 00000000..42d37377 --- /dev/null +++ b/.github/scripts/tests/test_notification_maintainer_followup.py @@ -0,0 +1,150 @@ +"""Maintainer acceptance at inert actual notification consumers.""" +import unittest +from notification_fixture import templates, receive, LANGUAGES, SCRIPTS, extract +from notification_final_fixture import deliver + +NATIVE_REPORT = '''Details +======= +VMID Name Status Time Size Filename +100 web ok 1m 1s 1 GiB vm/100/2026-09-29T17:00:00Z + +Total running time: 1m 1s +Total size: 1 GiB + +Logs +==== +vzdump --all 1 --storage PBS --mode snapshot + +100: 2026-09-29 17:00:00 INFO: Starting Backup of VM 100 (qemu) +100: 2026-09-29 17:01:01 INFO: Finished Backup of VM 100 (00:01:01) +''' + +class MaintainerFollowupTests(unittest.TestCase): + def test_completed_with_warnings_requires_independent_completion(self): + warning = '\n100: 2026-09-29 17:00:01 WARN: file changed during backup' + event = receive(NATIVE_REPORT + warning) + self.assertEqual(event.data['backup_outcome'], 'completed_with_warnings') + self.assertEqual((event.event_type,event.severity), ('backup_complete','INFO')) + for manual in (False,True): + for lang in LANGUAGES: + result = deliver(event.event_type,event.data,event.severity,lang,manual=manual) + label = templates.runtime_message('backup.warningTitle',lang,hostname='node-a') + self.assertIn(label,result['title']) + status = templates.runtime_message('channels.email.status.completed_with_warnings',lang) + self.assertIn(status,result['text']) + self.assertIn('file changed during backup',result['text']) + # Manual sends intentionally skip channel emoji enrichment. + if not manual: self.assertTrue(result['title'].startswith('💾⚠️')) + rich, _ = templates.enrich_with_emojis(event.event_type,result['title'],result['body'],event.data) + self.assertTrue(rich.startswith('💾⚠️')) + self.assertEqual(receive('WARN: file changed during backup').data['backup_outcome'],'unconfirmed') + self.assertEqual(receive(NATIVE_REPORT.split('Total running time:')[0]+warning).data['backup_outcome'],'unconfirmed') + self.assertEqual(receive(NATIVE_REPORT+warning+'\nERROR: cleanup failed').data['backup_outcome'],'failed') + self.assertEqual(receive(NATIVE_REPORT+warning,'warning').data['backup_outcome'],'completed_with_warnings') + self.assertEqual(receive('INFO: Starting Backup of VM 100 (qemu)\nINFO: Finished Backup of VM 100 (00:01:01)'+warning).data['backup_outcome'],'completed_with_warnings') + self.assertEqual(receive(NATIVE_REPORT+warning,kind='').data['backup_outcome'],'unconfirmed') + + def test_null_filename_failed_guest_uses_own_start_identity(self): + report = NATIVE_REPORT.replace('100 web ok 1m 1s 1 GiB vm/100/2026-09-29T17:00:00Z', + '100 web err 1m 1s 0 B null') + for kind, prefix in (('qemu','VM'),('lxc','CT')): + own = report.replace('VM 100 (qemu)',f'VM 100 ({kind})') + # Unrelated guest appears first and must never supply the failed type. + message = f'INFO: Starting Backup of VM 999 ({"lxc" if kind == "qemu" else "qemu"})\n' + own + event = receive(message,'error','vzdump backup status (raw-node): backup failed') + parsed = templates._parse_vzdump_message(message) + self.assertEqual(parsed['vms'][0]['type'],kind) + result = deliver(event.event_type,event.data,event.severity) + self.assertIn(f'{prefix} web (100)',result['title']) + self.assertIn(f'❌ {prefix} web (100)',result['body']) + for message in (report.replace('Starting Backup of VM 100','Starting Backup of VM 999'), + report+'\nINFO: Starting Backup of VM 100 (lxc)'): + self.assertEqual(templates._parse_vzdump_message(message)['vms'][0]['type'],'') + + def test_original_subject_only_retained_when_no_guest_context(self): + subject = 'vzdump backup status (raw-host): backup failed: multiple problems' + event = receive('ERROR: archive write failed\n'+NATIVE_REPORT,'error',subject) + event.data['hostname']='configured-alias' + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang) + self.assertNotIn(subject,result['text']) + self.assertEqual(result['text'].count('ERROR: archive write failed'),1) + self.assertIn('configured-alias',result['title']) + setup=receive('Details\n=======\nVMID Name Status Time Size Filename\n\nTotal running time: 0s\nTotal size: 0 B','error',subject.replace('multiple problems','unable to open storage')) + result=deliver(setup.event_type,setup.data,setup.severity) + self.assertEqual(result['text'].count(setup.data['pve_title']),1) + unique=receive(NATIVE_REPORT,'error',subject.replace('multiple problems','job-end hook denied')) + result=deliver(unique.event_type,unique.data,unique.severity) + self.assertIn('job-end hook denied',result['body']) + self.assertNotIn('vzdump backup status',result['body']) + + def test_backup_diagnostics_are_bounded_with_principal_cause_and_notice(self): + raw = NATIVE_REPORT + '\n' + '\n'.join(f'WARN: repeated warning {i}' for i in range(80)) + '\nERROR: principal archive write failure' + event=receive(raw) + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang) + self.assertIn('ERROR: principal archive write failure',result['body']) + self.assertLessEqual(len('\n'.join(line for line in result['body'].splitlines() if line.startswith(('WARN:','ERROR:')))),1024) + self.assertLessEqual(sum(line.startswith(('WARN:','ERROR:')) for line in result['body'].splitlines()),8) + notice=templates.runtime_message('backup.diagnosticsOmitted',lang,count=73) + self.assertIn(notice,result['body']) + self.assertEqual(event.data['pve_message'],raw) + long=receive('ERROR: '+ 'b'*5000,'error','vzdump backup status (node): backup failed') + result=deliver(long.event_type,long.data,long.severity) + self.assertLess(len(result['body']),1400) + self.assertIn('ERROR: '+ 'b'*100,result['body']) + self.assertIn(templates.runtime_message('backup.diagnosticsOmitted','en',count=1),result['body']) + + def test_real_quiet_digest_retains_each_backup_outcome_icon(self): + samples=[(NATIVE_REPORT,'confirmed','💾✅'),(NATIVE_REPORT+'\nWARN: changed file','completed_with_warnings','💾⚠️'),(NATIVE_REPORT+'\nERROR: write failed','failed','💾❌'),('INFO: Starting Backup of VM 100 (qemu)','unconfirmed','💾❔')] + for raw,outcome,icon in samples: + event=receive(raw) + self.assertEqual(event.data['backup_outcome'],outcome) + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang,quiet=True) + self.assertIn(icon,result['body']) + self.assertNotIn('💾❔',result['body']) if outcome!='unconfirmed' else None + import types,datetime + ns={'datetime':datetime.datetime,'runtime_message':templates.runtime_message,'EVENT_EMOJI':templates.EVENT_EMOJI,'CATEGORY_EMOJI':templates.CATEGORY_EMOJI} + compose=extract(SCRIPTS/'notification_manager.py','_compose_digest_body','NotificationManager',ns) + target=types.SimpleNamespace(_notification_language=lambda:'en') + rows=[(i,'backup_complete','backup',1,icon+' node: Backup','') for i,(_,_,icon) in enumerate(samples)] + body=compose(target,rows,use_icons=True) + for _,_,icon in samples:self.assertIn(icon,body) + plain=compose(target,rows,use_icons=False) + for _,_,icon in samples:self.assertNotIn(icon,plain) + should=extract(SCRIPTS/'notification_manager.py','_should_buffer_for_digest','NotificationManager',{}) + self.assertFalse(should(types.SimpleNamespace(_DIGEST_EXEMPT_EVENTS={'backup_complete'},_config={'email.digest_enabled':'true'}),'email','INFO','backup_complete')) + + def test_quiet_restore_details_are_subordinate_in_text_and_email(self): + from notification_final_fixture import restore_event + event=restore_event('missing module zfs') + for lang in LANGUAGES: + result=deliver(event['event_type'],event['data'],event['severity'],lang,quiet=True) + body_lines=result['buffered'][0][3].splitlines() + for line in body_lines: + if line.strip(): + self.assertIn(' '+line.strip(),result['body']) + self.assertIn(' '+__import__('html').escape(line.strip()),result['html']) + self.assertIn('white-space:pre-wrap;',result['html']) + self.assertNotIn(templates.runtime_message('digest.footer',lang),result['body']) + + def test_job_level_subject_cause_survives_warning_cap(self): + raw=NATIVE_REPORT+'\n'+'\n'.join('WARN: repeated diagnostic '+str(i) for i in range(80)) + event=receive(raw,'error','vzdump backup status (raw-host): backup failed: job-end hook denied') + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang) + self.assertEqual(result['body'].count('job-end hook denied'),1) + self.assertNotIn('vzdump backup status',result['body']) + self.assertIn(templates.runtime_message('backup.diagnosticsOmitted',lang,count=73),result['body']) + + def test_backup_quiet_email_uses_event_scoped_wrapping(self): + event=receive(NATIVE_REPORT+'\nWARN: changed file') + result=deliver(event.event_type,event.data,event.severity,'sv',quiet=True) + self.assertTrue(result['data'].get('_backup_summary')) + self.assertIn('table-layout:fixed;',result['html']) + self.assertIn('overflow-wrap:break-word;',result['html']) + unrelated=deliver('node_reconnect',{'hostname':'node-a'},'OK') + self.assertNotIn('table-layout:fixed;',unrelated['html']) + +if __name__ == '__main__': unittest.main() diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py index 5be667c3..9dabb21f 100644 --- a/.github/scripts/tests/test_notification_outcome_wording.py +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -134,7 +134,7 @@ class OutcomeWording(unittest.TestCase): ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\n'+header+'\n'+row_ok+'\nTotal running time: 00:01:00\n'+truncated, 'confirmed'), ('vzdump', 'info', header+'\n'+row_ok+'\nTotal running time: 00:01:00\n'+truncated, 'confirmed'), ('vzdump', 'info', header+'\n'+row_err+'\nTotal running time: 00:01:00\n'+truncated, 'failed'), - ('vzdump', 'warning', header+'\n'+row_ok+'\nTotal running time: 00:01:00', 'unconfirmed'), + ('vzdump', 'warning', header+'\n'+row_ok+'\nTotal running time: 00:01:00', 'completed_with_warnings'), ('vzdump', 'info', header+'\n'+row_ok, 'unconfirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\n'+header+'\n'+row_ok, 'unconfirmed'), ('vzdump', 'warning', header+'\n'+row_err+'\nTotal running time: 00:01:00', 'failed'), @@ -147,11 +147,11 @@ class OutcomeWording(unittest.TestCase): ('vzdump', 'info', header+'\n'+row_ok+'\n'+row_warning+'\nTotal running time: 00:02:00', 'unconfirmed'), ('vzdump', 'info', header+'\n'+row_ok[:30], 'unconfirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'confirmed'), - ('vzdump', 'warning', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'unconfirmed'), + ('vzdump', 'warning', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'completed_with_warnings'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)', 'unconfirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Starting Backup of VM 105 (lxc)\nINFO: Finished Backup of VM 105 (00:01:00)', 'unconfirmed'), - ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nWARNING: skipped file', 'unconfirmed'), - ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: TASK OK\n104 alpha WARNINGS: 1', 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nWARNING: skipped file', 'completed_with_warnings'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: TASK OK\n104 alpha WARNINGS: 1', 'completed_with_warnings'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 105 (00:01:00)', 'unconfirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Starting Backup of VM 105 (lxc)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: Finished Backup of VM 105 (00:01:00)', 'confirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nERROR: backup failed for VM 104', 'failed'), diff --git a/.github/scripts/tests/test_notification_pve92.py b/.github/scripts/tests/test_notification_pve92.py index 16f27487..220d129a 100644 --- a/.github/scripts/tests/test_notification_pve92.py +++ b/.github/scripts/tests/test_notification_pve92.py @@ -128,10 +128,10 @@ class PVE92Tests(unittest.TestCase): self.assertEqual(event_for(message).data['backup_outcome'], 'confirmed') for message, severity, expected in ( (PVE92.replace('ok ', 'OK '), 'info', 'confirmed'), - (message, 'warning', 'unconfirmed'), + (message, 'warning', 'completed_with_warnings'), (message, 'error', 'failed'), (message + '\nERROR: archive write failed', 'info', 'failed'), - (message + '\nWARNING: skipped file', 'info', 'unconfirmed'), + (message + '\nWARNING: skipped file', 'info', 'completed_with_warnings'), (PVE92.replace('ok ', 'WARNINGS '), 'info', 'unconfirmed'), ): with self.subTest(message=message, severity=severity): diff --git a/.github/scripts/tests/test_notification_recovery_evidence.py b/.github/scripts/tests/test_notification_recovery_evidence.py new file mode 100644 index 00000000..45f316f8 --- /dev/null +++ b/.github/scripts/tests/test_notification_recovery_evidence.py @@ -0,0 +1,107 @@ +"""Fresh existing-check provenance; extracted consumers, real disposable SQLite.""" +import contextlib +import datetime +import json +import os +import sqlite3 +import tempfile +import time +import types +import unittest +from unittest.mock import patch +from notification_fixture import extract, SCRIPTS, templates, LANGUAGES +from notification_final_fixture import deliver + +class RecoveryEvidenceTests(unittest.TestCase): + def test_cpu_success_provenance_is_persisted_only_after_normal_samples(self): + events=[] + with tempfile.TemporaryDirectory() as scratch: + db=scratch+'/health.sqlite' + conn=sqlite3.connect(db) + conn.execute('CREATE TABLE errors(id INTEGER PRIMARY KEY,error_key TEXT,details TEXT,resolved_at TEXT,resolution_type TEXT,resolution_reason TEXT)') + conn.execute("INSERT INTO errors(error_key,details) VALUES ('cpu_usage','{}')") + conn.commit();conn.close() + @contextlib.contextmanager + def connection(): + c=sqlite3.connect(db) + try:yield c + finally:c.close() + ns={'datetime':datetime.datetime,'json':json} + resolve=extract(SCRIPTS/'health_persistence.py','_resolve_error_impl','HealthPersistence',ns) + store=types.SimpleNamespace(_db_connection=connection,_entity_from_details=lambda details:'',_record_event=lambda cursor,kind,key,data:events.append(data)) + store.resolve_error=lambda key,reason,**kw:resolve(store,key,reason,**kw) + ns={'Dict':dict,'Any':object,'os':os,'time':time,'health_persistence':store,'psutil':types.SimpleNamespace(cpu_percent=lambda **kw:20,cpu_count=lambda:4)} + check=extract(SCRIPTS/'health_monitor.py','_check_cpu_with_hysteresis','HealthMonitor',ns) + target=types.SimpleNamespace(state_history={'cpu_usage':[{'value':20,'time':time.time()-i*10} for i in range(10)]},CPU_CRITICAL=95,CPU_WARNING=85,CPU_RECOVERY=75,CPU_CRITICAL_DURATION=300,CPU_WARNING_DURATION=300,CPU_RECOVERY_DURATION=120,_check_cpu_temperature=lambda:None) + result=check(target) + self.assertEqual(result['status'],'OK') + self.assertTrue(events[-1].get('check_evidence'),events) + proof=events[-1]['check_evidence'] + self.assertEqual(proof['check'],'cpu_usage') + self.assertGreaterEqual(proof['checked_at'],time.time()-5) + # Existing generic resolve callers (cleanup/exclusion) get no proof. + conn=sqlite3.connect(db);conn.execute('UPDATE errors SET resolved_at=NULL');conn.commit();conn.close() + resolve(store,'cpu_usage','No longer present') + self.assertFalse(events[-1].get('check_evidence')) + + def test_recovery_query_requires_fresh_same_incident_proof(self): + with tempfile.TemporaryDirectory() as scratch: + db=scratch+'/health.sqlite' + conn=sqlite3.connect(db) + conn.execute('CREATE TABLE errors(id INTEGER PRIMARY KEY,error_key TEXT,first_seen TEXT,last_seen TEXT,resolved_at TEXT,acknowledged INTEGER)') + conn.execute('CREATE TABLE events(id INTEGER PRIMARY KEY,event_type TEXT,error_key TEXT,timestamp TEXT,data TEXT)') + now=datetime.datetime.now(); first=(now-datetime.timedelta(minutes=10)).isoformat(); last=(now-datetime.timedelta(minutes=1)).isoformat(); resolved=now.isoformat() + proof={'check':'cpu_usage','checked_at':now.timestamp()} + conn.execute('INSERT INTO errors VALUES(1,?,?,?,?,0)',('cpu_usage',first,last,resolved)) + conn.execute('INSERT INTO events VALUES(1,?,?,?,?)',('resolved','cpu_usage',resolved,json.dumps({'check_evidence':proof}))) + conn.commit();conn.close() + @contextlib.contextmanager + def connection(**kwargs): + c=sqlite3.connect(db) + try:yield c + finally:c.close() + ns={'datetime':datetime.datetime,'json':json,'time':time} + tree=(SCRIPTS/'health_persistence.py').read_text() + query=extract(SCRIPTS/'health_persistence.py','get_recovery_evidence','HealthPersistence',ns) if 'def get_recovery_evidence(' in tree else lambda *args:None + store=types.SimpleNamespace(_db_connection=connection) + self.assertEqual(query(store,'cpu_usage',first),proof) + self.assertIsNone(query(store,'cpu_usage','different incident')) + for field,value in [('acknowledged',1),('resolved_at',None),('last_seen',(now+datetime.timedelta(seconds=1)).isoformat())]: + conn=sqlite3.connect(db);conn.execute(f'UPDATE errors SET {field}=?',(value,));conn.commit();conn.close() + self.assertIsNone(query(store,'cpu_usage',first)) + conn=sqlite3.connect(db);conn.execute('UPDATE errors SET acknowledged=0,resolved_at=?,last_seen=?',(resolved,last));conn.commit();conn.close() + for bad in (None,{'check':'cpu_usage','checked_at':now.timestamp()-7201},{'check':'storage_removed','checked_at':now.timestamp()},{'check':'cpu_usage','checked_at':float('inf')},{'check':'cpu_usage','checked_at':10**400}): + conn=sqlite3.connect(db);conn.execute('UPDATE events SET data=?',(json.dumps({'check_evidence':bad}),));conn.commit();conn.close() + self.assertIsNone(query(store,'cpu_usage',first)) + + def test_poller_and_all_consumers_distinguish_proven_recovery_from_disappearance(self): + import sys + ns={'time':time,'json':json,'Dict':dict,'NotificationEvent':lambda *a,**kw:types.SimpleNamespace(event_type=a[0],severity=a[1],data=a[2])} + poll=extract(SCRIPTS/'notification_events.py','_check_persistent_health','PollingCollector',ns) + for proof in (None,{'check':'cpu_usage','checked_at':time.time()}): + events=[] + store=types.SimpleNamespace(get_active_errors=lambda:[],is_error_acknowledged=lambda key:False,get_recovery_evidence=lambda *a:proof) + collector=types.SimpleNamespace(_hostname='node-a',_ENTITY_MAP={'cpu':('node','')},_first_poll_done=True,_known_errors={'cpu_usage':{'category':'cpu','reason':'CPU high','severity':'WARNING','first_seen':'2026-09-30T00:00:00'}},_notified_severity={'cpu_usage':'WARNING'},_last_notified={'cpu_usage':1},_queue=types.SimpleNamespace(put=events.append),_guest_storage_error_is_now_foreign=lambda *a:False,_save_known_errors_meta=lambda:None) + with patch.dict(sys.modules,{'health_persistence':types.SimpleNamespace(health_persistence=store)}):poll(collector) + self.assertEqual(len(events),1) + event=events[0] + self.assertEqual(event.data.get('recovery_outcome'),'resolved' if proof else 'no_longer_reported') + self.assertEqual(event.data['is_recovery'],bool(proof)) + for lang in LANGUAGES: + for manual in (False,True): + result=deliver(event.event_type,event.data,event.severity,lang,manual=manual) + if proof: + self.assertIn(templates.runtime_message('healthRecovery.title',lang,hostname='node-a',category='cpu',entity_suffix=''),result['title']) + self.assertIn('background:#f0fdf4;',result['html']) + else:self.assertNotIn('background:#f0fdf4;',result['html']) + self.assertEqual(result['text'].count(event.data['reason']),1) + + def test_manual_recovery_flag_alone_is_not_authoritative_evidence(self): + data={'hostname':'node-a','category':'cpu','reason':'Observation disappeared','duration':'1h','original_severity':'WARNING','recovery_outcome':'resolved'} + for lang in LANGUAGES: + for manual in (False,True): + result=deliver('error_resolved',data,'OK',lang,manual=manual) + self.assertNotIn('background:#f0fdf4;',result['html']) + self.assertNotIn(templates.runtime_message('healthRecovery.body',lang,**data),result['body']) + +if __name__=='__main__':unittest.main() diff --git a/AppImage/messages/de/common.json b/AppImage/messages/de/common.json index 0cb5b7c0..424c5aba 100644 --- a/AppImage/messages/de/common.json +++ b/AppImage/messages/de/common.json @@ -6980,7 +6980,8 @@ "failed": "Fehlgeschlagen", "completed": "Abgeschlossen", "started": "Gestartet", - "unconfirmed": "Nicht bestätigt" + "unconfirmed": "Nicht bestätigt", + "completed_with_warnings": "Mit Warnungen abgeschlossen" }, "report": "{group} Bericht", "details": "Details", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "Die hohen Messwerte erstrecken sich über {duration}." }, + "healthRecovery": {"title": "{hostname}: Behoben - {category}{entity_suffix}", "body": "Eine aktuelle Zustandsprüfung hat für {category} wieder einen normalen Zustand festgestellt.\nVorherige Beobachtung: {reason}\nVorheriger Schweregrad: {original_severity}\nZeit seit der ersten Beobachtung: {duration}", "status": "Behoben"}, "backup": { "confirmedTitle": "{hostname}: Backup abgeschlossen", "confirmedBody": "Backup erfolgreich abgeschlossen.", "errorTitle": "{hostname}: Backup-Fehler gemeldet", "errorBody": "Der Backup-Bericht enthält einen Fehler.", - "unconfirmedBody": "Das Backup-Ergebnis ist nicht bestätigt." + "unconfirmedBody": "Das Backup-Ergebnis ist nicht bestätigt.", + "warningTitle": "{hostname}: Sicherung mit Warnungen abgeschlossen", + "warningBody": "Sicherung mit Warnungen abgeschlossen.", + "diagnosticsOmitted": "Weitere Diagnosezeilen oder Text ausgelassen: {count}. Originalbericht bleibt erhalten." } } } diff --git a/AppImage/messages/en/common.json b/AppImage/messages/en/common.json index abd42a64..7d0ca28a 100644 --- a/AppImage/messages/en/common.json +++ b/AppImage/messages/en/common.json @@ -6251,5 +6251,5 @@ "cancel": "Cancel" } }, - "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: No longer reported - {category}{entity_suffix}","body":"The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname}: Backup outcome unconfirmed","body":"The backup outcome could not be confirmed from this notice.","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or run the post-install menu (option 2) → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Settings > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice","observation":"No longer reported"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"backup":{"confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed."}}} + "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: No longer reported - {category}{entity_suffix}","body":"The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname}: Backup outcome unconfirmed","body":"The backup outcome could not be confirmed from this notice.","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or run the post-install menu (option 2) → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Settings > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice","observation":"No longer reported"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed","completed_with_warnings":"Completed with warnings"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"healthRecovery":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"The {category} condition returned to normal in a fresh health check.\nPrevious observation: {reason}\nPrevious severity: {original_severity}\nTime since first observation: {duration}","status":"Resolved"},"backup":{"confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed.","warningTitle":"{hostname}: Backup completed with warnings","warningBody":"Backup completed with warnings.","diagnosticsOmitted":"Additional diagnostic lines or text omitted: {count}. Original report retained."}}} } diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index e25e0a1a..961a228e 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -6980,7 +6980,8 @@ "failed": "Fallido", "completed": "Completado", "started": "Iniciado", - "unconfirmed": "Sin confirmar" + "unconfirmed": "Sin confirmar", + "completed_with_warnings": "Completado con advertencias" }, "report": "Informe de {group}", "details": "Detalles", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "Lecturas altas registradas a lo largo de {duration}." }, + "healthRecovery": {"title": "{hostname}: Resuelto - {category}{entity_suffix}", "body": "Una comprobación reciente confirma que la condición de {category} volvió a la normalidad.\nObservación anterior: {reason}\nGravedad anterior: {original_severity}\nTiempo desde la primera observación: {duration}", "status": "Resuelto"}, "backup": { "confirmedTitle": "{hostname}: backup completado", "confirmedBody": "Backup completado correctamente.", "errorTitle": "{hostname}: error notificado en el backup", "errorBody": "El informe del backup contiene un error.", - "unconfirmedBody": "El resultado del backup no está confirmado." + "unconfirmedBody": "El resultado del backup no está confirmado.", + "warningTitle": "{hostname}: Backup completado con advertencias", + "warningBody": "Backup completado con advertencias.", + "diagnosticsOmitted": "Líneas o texto de diagnóstico omitidos: {count}. Se conserva el informe original." } } } diff --git a/AppImage/messages/fr/common.json b/AppImage/messages/fr/common.json index d3ebeea9..a0ffa399 100644 --- a/AppImage/messages/fr/common.json +++ b/AppImage/messages/fr/common.json @@ -6980,7 +6980,8 @@ "failed": "Échec", "completed": "Terminé", "started": "Commencé", - "unconfirmed": "Non confirmé" + "unconfirmed": "Non confirmé", + "completed_with_warnings": "Terminée avec avertissements" }, "report": "Rapport {group}", "details": "Détails", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "Les relevés élevés s'étendent sur {duration}." }, + "healthRecovery": {"title": "{hostname} : Résolu - {category}{entity_suffix}", "body": "Un contrôle récent confirme le retour à la normale de la condition {category}.\nObservation précédente : {reason}\nGravité précédente : {original_severity}\nTemps depuis la première observation : {duration}", "status": "Résolu"}, "backup": { "confirmedTitle": "{hostname} : sauvegarde terminée", "confirmedBody": "Sauvegarde terminée avec succès.", "errorTitle": "{hostname} : erreur signalée lors de la sauvegarde", "errorBody": "Le rapport de sauvegarde contient une erreur.", - "unconfirmedBody": "Le résultat de la sauvegarde n’est pas confirmé." + "unconfirmedBody": "Le résultat de la sauvegarde n’est pas confirmé.", + "warningTitle": "{hostname}: Sauvegarde terminée avec avertissements", + "warningBody": "Sauvegarde terminée avec avertissements.", + "diagnosticsOmitted": "Lignes ou texte de diagnostic omis : {count}. Rapport original conservé." } } } diff --git a/AppImage/messages/it/common.json b/AppImage/messages/it/common.json index 2b7f0990..456091d1 100644 --- a/AppImage/messages/it/common.json +++ b/AppImage/messages/it/common.json @@ -6980,7 +6980,8 @@ "failed": "Fallito", "completed": "Completato", "started": "Iniziato", - "unconfirmed": "Non confermato" + "unconfirmed": "Non confermato", + "completed_with_warnings": "Completato con avvisi" }, "report": "{group} Rapporto", "details": "Dettagli", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "Intervallo dei campioni sopra soglia: {duration}." }, + "healthRecovery": {"title": "{hostname}: Risolto - {category}{entity_suffix}", "body": "Un controllo recente conferma che la condizione {category} è tornata nella norma.\nOsservazione precedente: {reason}\nGravità precedente: {original_severity}\nTempo dalla prima osservazione: {duration}", "status": "Risolto"}, "backup": { "confirmedTitle": "{hostname}: backup completato", "confirmedBody": "Backup completato correttamente.", "errorTitle": "{hostname}: errore segnalato nel backup", "errorBody": "Il rapporto del backup contiene un errore.", - "unconfirmedBody": "L’esito del backup non è confermato." + "unconfirmedBody": "L’esito del backup non è confermato.", + "warningTitle": "{hostname}: Backup completato con avvisi", + "warningBody": "Backup completato con avvisi.", + "diagnosticsOmitted": "Righe o testo diagnostico omessi: {count}. Il report originale è conservato." } } } diff --git a/AppImage/messages/pt/common.json b/AppImage/messages/pt/common.json index 98580f03..6b731ed2 100644 --- a/AppImage/messages/pt/common.json +++ b/AppImage/messages/pt/common.json @@ -6980,7 +6980,8 @@ "failed": "Falhou", "completed": "Concluído", "started": "Iniciado", - "unconfirmed": "Não confirmado" + "unconfirmed": "Não confirmado", + "completed_with_warnings": "Concluído com avisos" }, "report": "Relatório {group}", "details": "Detalhes", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "As amostras elevadas abrangem {duration}." }, + "healthRecovery": {"title": "{hostname}: Resolvido - {category}{entity_suffix}", "body": "Uma verificação recente confirma que a condição de {category} voltou ao normal.\nObservação anterior: {reason}\nGravidade anterior: {original_severity}\nTempo desde a primeira observação: {duration}", "status": "Resolvido"}, "backup": { "confirmedTitle": "{hostname}: backup concluído", "confirmedBody": "Backup concluído com sucesso.", "errorTitle": "{hostname}: erro comunicado no backup", "errorBody": "O relatório do backup contém um erro.", - "unconfirmedBody": "O resultado do backup não está confirmado." + "unconfirmedBody": "O resultado do backup não está confirmado.", + "warningTitle": "{hostname}: Backup concluído com avisos", + "warningBody": "Backup concluído com avisos.", + "diagnosticsOmitted": "Linhas ou texto de diagnóstico omitidos: {count}. Relatório original preservado." } } } diff --git a/AppImage/messages/sv/common.json b/AppImage/messages/sv/common.json index 405e1553..d1842183 100644 --- a/AppImage/messages/sv/common.json +++ b/AppImage/messages/sv/common.json @@ -6980,7 +6980,8 @@ "failed": "Misslyckades", "completed": "Klar", "started": "Startat", - "unconfirmed": "Obekräftat" + "unconfirmed": "Obekräftat", + "completed_with_warnings": "Klar med varningar" }, "report": "{group} Rapportera", "details": "Detaljer", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "De höga mätvärdena sträcker sig över {duration}." }, + "healthRecovery": {"title": "{hostname}: Åtgärdat - {category}{entity_suffix}", "body": "En aktuell hälsokontroll bekräftar att tillståndet för {category} återgått till det normala.\nTidigare observation: {reason}\nTidigare allvarlighetsgrad: {original_severity}\nTid sedan första observationen: {duration}", "status": "Åtgärdat"}, "backup": { "confirmedTitle": "{hostname}: säkerhetskopiering klar", "confirmedBody": "Säkerhetskopieringen slutfördes utan fel.", "errorTitle": "{hostname}: fel rapporterat vid säkerhetskopiering", "errorBody": "Rapporten om säkerhetskopieringen innehåller ett fel.", - "unconfirmedBody": "Säkerhetskopieringens resultat är inte bekräftat." + "unconfirmedBody": "Säkerhetskopieringens resultat är inte bekräftat.", + "warningTitle": "{hostname}: Säkerhetskopiering klar med varningar", + "warningBody": "Säkerhetskopiering klar med varningar.", + "diagnosticsOmitted": "Utelämnade diagnosrader eller text: {count}. Originalrapporten bevaras." } } } diff --git a/AppImage/scripts/health_monitor.py b/AppImage/scripts/health_monitor.py index 26d38470..565f68e3 100644 --- a/AppImage/scripts/health_monitor.py +++ b/AppImage/scripts/health_monitor.py @@ -1452,7 +1452,11 @@ class HealthMonitor: status = 'OK' reason = None # CPU is normal - auto-resolve any existing CPU errors - health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal') + evidence = None + if cpu_percent < self.CPU_RECOVERY and len(recovery_samples) >= RECOVERY_MIN_SAMPLES: + evidence = {'check': 'cpu_usage', 'checked_at': current_time} + health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal', + check_evidence=evidence) temp_status = self._check_cpu_temperature() diff --git a/AppImage/scripts/health_persistence.py b/AppImage/scripts/health_persistence.py index 4fb13fc7..35458418 100644 --- a/AppImage/scripts/health_persistence.py +++ b/AppImage/scripts/health_persistence.py @@ -744,12 +744,12 @@ class HealthPersistence: return event_info - def resolve_error(self, error_key: str, reason: str = 'auto-resolved'): + def resolve_error(self, error_key: str, reason: str = 'auto-resolved', *, check_evidence=None): """Mark an error as resolved""" with self._db_lock: - return self._resolve_error_impl(error_key, reason) + return self._resolve_error_impl(error_key, reason, check_evidence=check_evidence) - def _resolve_error_impl(self, error_key, reason): + def _resolve_error_impl(self, error_key, reason, *, check_evidence=None): with self._db_connection() as conn: cursor = conn.cursor() now = datetime.now().isoformat() @@ -788,12 +788,59 @@ class HealthPersistence: stored_details = None self._record_event(cursor, 'resolved', error_key, { 'reason': reason, + # Only explicit current-check callers attach this proof. + # Generic resolve/cleanup remains neutral. + 'check_evidence': check_evidence, 'entity': self._entity_from_details(stored_details), 'details': stored_details or {}, }) conn.commit() + def get_recovery_evidence(self, error_key: str, first_seen: str): + """Return fresh same-incident native check proof, never absence of errors. + + Initially only the host CPU check has a stable condition identity. Other + checks, generic clears, excluded/deleted records and legacy events stay + neutral until they have equivalent per-condition provenance. + """ + if error_key != 'cpu_usage' or not first_seen: + return None + try: + with self._db_connection() as conn: + row = conn.execute(''' + SELECT first_seen, last_seen, resolved_at, acknowledged + FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1 + ''', (error_key,)).fetchone() + if not row or row[0] != first_seen or not row[2] or row[3]: + return None + event = conn.execute(''' + SELECT timestamp, data FROM events + WHERE error_key = ? AND event_type = 'resolved' + ORDER BY id DESC LIMIT 1 + ''', (error_key,)).fetchone() + if not event: + return None + proof = json.loads(event[1]).get('check_evidence') + if not isinstance(proof, dict) or proof.get('check') != error_key: + return None + checked = proof.get('checked_at') + if not isinstance(checked, (int, float)) or isinstance(checked, bool): + return None + checked = float(checked) + # Reuse the collector's existing two-hour freshness boundary. + now = datetime.now().timestamp() + if not 0 <= now - checked <= 7200: + return None + last_seen = datetime.fromisoformat(row[1]).timestamp() + resolved = datetime.fromisoformat(row[2]).timestamp() + recorded = datetime.fromisoformat(event[0]).timestamp() + if not last_seen <= checked <= resolved <= recorded: + return None + return proof + except (ValueError, TypeError, AttributeError, OverflowError): + return None + def is_error_active(self, error_key: str, category: Optional[str] = None) -> bool: """ Check if an error is currently active OR suppressed (dismissed but within suppression period). diff --git a/AppImage/scripts/notification_channels.py b/AppImage/scripts/notification_channels.py index bf74ddeb..b0b106ab 100644 --- a/AppImage/scripts/notification_channels.py +++ b/AppImage/scripts/notification_channels.py @@ -1038,13 +1038,23 @@ class EmailChannel(NotificationChannel): # Determine group for section header event_type = data.get('_event_type', '') if event_type == 'error_resolved': - sev.update(self._SEV_DEFAULT) - sev['label'] = _runtime_text('email.severity.observation', data) + if (data.get('recovery_outcome') == 'resolved' + and data.get('is_recovery') is True + and isinstance(data.get('check_evidence'), dict) + and data['check_evidence'].get('check') == 'cpu_usage'): + sev.update(self._SEV_STYLE['OK']) + sev['label'] = _runtime_notification_text('healthRecovery.status', data) + else: + sev.update(self._SEV_DEFAULT) + sev['label'] = _runtime_text('email.severity.observation', data) elif event_type == 'backup_complete': outcome = data.get('backup_outcome') if outcome == 'confirmed': sev.update(self._SEV_STYLE['OK']) status = 'completed' + elif outcome == 'completed_with_warnings': + sev.update(self._SEV_STYLE['WARNING']) + status = 'completed_with_warnings' elif outcome == 'failed': sev.update(self._SEV_STYLE['CRITICAL']) status = 'failed' @@ -1057,7 +1067,7 @@ class EmailChannel(NotificationChannel): # bodies, including restore bodies released from quiet hours. backup_email = event_type in {'backup_complete', 'backup_fail'} wrap_body = (event_type in {'temp_high', 'system_restore_completed', 'error_resolved'} - or backup_email or data.get('_restore_summary')) + or backup_email or data.get('_restore_summary') or data.get('_backup_summary')) temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else '' temp_table_layout = 'table-layout:fixed;' if wrap_body else '' backup_title_wrap = temp_cell_wrap if backup_email else '' @@ -1103,7 +1113,7 @@ class EmailChannel(NotificationChannel): # Observation age/disappearance must not become a green OK row. # The endpoint's warnings_block and task counts live in the # localized body, not the generic services Event row. - detail_rows = [('', html_mod.escape(line.strip())) + detail_rows = [('', html_mod.escape(line if data.get('_quiet_hours_summary') else line.strip())) for line in body.split('\n') if line.strip()] # ── Fallback: if no structured rows, render body text lines ── @@ -1122,6 +1132,7 @@ class EmailChannel(NotificationChannel): # ── Render detail rows as HTML table ── rows_html = '' + summary_whitespace = 'white-space:pre-wrap;' if data.get('_quiet_hours_summary') and data.get('_restore_summary') else '' for label, value in detail_rows: if label: rows_html += f'''