From dcf00cc6b92ee67cd7f3d116607edcb00326d848 Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Tue, 29 Sep 2026 09:56:28 +0200 Subject: [PATCH 01/14] fix(monitor): distinguish notification recovery and backup outcomes --- .../test_notification_outcome_wording.py | 391 ++++++++++++++++++ AppImage/messages/de/common.json | 23 +- AppImage/messages/en/common.json | 2 +- AppImage/messages/es/common.json | 23 +- AppImage/messages/fr/common.json | 23 +- AppImage/messages/it/common.json | 23 +- AppImage/messages/pt/common.json | 23 +- AppImage/messages/sv/common.json | 23 +- AppImage/scripts/notification_channels.py | 27 +- AppImage/scripts/notification_events.py | 60 ++- AppImage/scripts/notification_templates.py | 65 +-- .../tests/test_notification_runtime_i18n.py | 4 +- .../tests/test_vzdump_webhook_truncation.py | 6 +- 13 files changed, 616 insertions(+), 77 deletions(-) create mode 100644 .github/scripts/tests/test_notification_outcome_wording.py diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py new file mode 100644 index 00000000..d94b33fa --- /dev/null +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -0,0 +1,391 @@ +"""Inert producer-to-renderer checks for notification outcome claims.""" +import ast +import copy +import json +import re +import time +import sys +import types +import unittest +from pathlib import Path +from typing import Any +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[3] +SCRIPTS = ROOT / 'AppImage/scripts' +CATALOG = ROOT / 'AppImage/messages/en/common.json' +EXPECTED = { + 'error_resolved': { + 'title': '{hostname}: No longer reported - {category}{entity_suffix}', + 'body': 'The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}', + 'label': 'Health issue no longer reported', + }, + + 'system_restore_completed': { + 'body': 'Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}', + }, +} + + +def extract(path, name, owner=None, namespace=None): + tree = ast.parse(path.read_text()) + nodes = tree.body + if owner: + nodes = next(n.body for n in nodes if isinstance(n, ast.ClassDef) and n.name == owner) + node = next(n for n in nodes if isinstance(n, ast.FunctionDef) and n.name == name) + node.decorator_list = [] + ns = namespace if namespace is not None else {} + exec(compile(ast.Module(body=[node], type_ignores=[]), str(path), 'exec'), ns) + return ns[name] + + +def renderer(catalog, translated=None): + path = SCRIPTS / 'notification_templates.py' + tree = ast.parse(path.read_text()) + templates = ast.literal_eval(next(n.value for n in tree.body if isinstance(n, ast.Assign) and any(isinstance(t, ast.Name) and t.id == 'TEMPLATES' for t in n.targets))) + def lookup(obj, key): + for part in key.split('.'): + obj = obj.get(part) if isinstance(obj, dict) else None + return obj + def message(key, language='en', **values): + source = translated if language == 'it' and translated is not None else catalog + namespace = source['runtime']['notifications'] + value = lookup(namespace, key) or lookup(catalog['runtime']['notifications'], key) or '' + return value.format_map(type('Safe', (dict,), {'__missing__': lambda self, k: ''})(values)) + ns = {'TEMPLATES': templates, 'Dict': dict, 'Any': Any, 'time': time, 're': re, + '_get_hostname': lambda: 'node-a', + '_load_runtime_catalog': lambda lang: (translated if lang == 'it' and translated is not None + else catalog)['runtime']['notifications'], + '_catalog_value': lookup, 'runtime_message': message} + from typing import Optional + ns['Optional'] = Optional + extract(path, '_parse_vzdump_message', namespace=ns) + extract(path, '_format_vzdump_body', namespace=ns) + return templates, extract(path, 'render_template', namespace=ns) + + +class OutcomeWording(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.catalog = json.loads(CATALOG.read_text()) + cls.templates, render = renderer(cls.catalog) + cls.render = staticmethod(render) + + def test_exact_four_english_leaves_match_source_and_catalog(self): + for event, fields in EXPECTED.items(): + for field, value in fields.items(): + with self.subTest(event=event, field=field): + self.assertEqual(self.templates[event][field], value) + self.assertEqual(self.catalog['runtime']['notifications']['templates'][event][field], value) + + def test_webhook_backup_outcome_is_evidence_based_without_rerouting(self): + path = SCRIPTS / 'notification_events.py' + ns = {'re': re, 'capture_journal_context': lambda **kw: ''} + classify = extract(path, '_classify_pve', 'ProxmoxHookWatcher', ns) + severity_map = extract(path, '_map_severity', 'ProxmoxHookWatcher', ns) + backup_outcome = extract(path, '_backup_outcome', 'ProxmoxHookWatcher', ns) + class Event: + def __init__(self, **kw): self.__dict__.update(kw); self.event_id = 'inert' + class Queue: + def __init__(self): self.items = [] + def put(self, event): self.items.append(event) + ns['NotificationEvent'] = Event + receive = extract(path, 'process_webhook', 'ProxmoxHookWatcher', ns) + class Receiver: + _hostname = 'node-a' + _classify_pve = classify + _map_severity = staticmethod(severity_map) + _backup_outcome = staticmethod(backup_outcome) + def __init__(self): self._queue = Queue() + header = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('VMID','Name','Status','Time','Size','Filename') + row_ok = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('104','alpha','OK','00:01:00','1.5 GiB','archive') + row_warning = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('105','beta','WARNINGS','00:01:00','1.5 GiB','archive') + row_error = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('105','beta','ERROR','00:01:00','1.5 GiB','archive') + cases = [ + ('vzdump', 'info', header+'\n'+row_ok+'\n'+row_error+'\nTotal running time: 00:02:00', 'failed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\n'+header+'\n'+row_error+'\nTotal running time: 00:02:00', 'failed'), + ('vzdump', 'info', header+'\n'+row_error, 'failed'), + ('vzdump', 'info', header+'\n'+row_ok+'\nTotal running time: 00:01:00', 'confirmed'), + ('vzdump', 'info', header+'\n'+row_ok+'\n'+row_warning+'\nTotal running time: 00:02:00', 'unconfirmed'), + ('vzdump', 'info', header+'\n'+row_ok[:30], 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'confirmed'), + ('vzdump', 'warning', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)', 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Starting Backup of VM 105 (lxc)\nINFO: Finished Backup of VM 105 (00:01:00)', 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nWARNING: skipped file', 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: TASK OK\n104 alpha WARNINGS: 1', 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 105 (00:01:00)', 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Starting Backup of VM 105 (lxc)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: Finished Backup of VM 105 (00:01:00)', 'confirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nERROR: backup failed for VM 104', 'failed'), + ('vzdump', 'warning', 'INFO: Starting Backup of VM 104 (qemu)\nERROR: backup failed', 'failed'), + ('', 'warning', 'Backup scheduled', 'unconfirmed'), + ('', 'info', 'Backup complete', 'unconfirmed'), + ] + for kind, severity, message, expected in cases: + with self.subTest(message=message, severity=severity): + receiver = Receiver() + receive(receiver, {'fields': {'type': kind}, 'severity': severity, + 'title': 'Backup', 'message': message}) + event = receiver._queue.items[0] + self.assertEqual(event.event_type, 'backup_complete') + self.assertEqual(event.data['backup_outcome'], expected) + + def test_backup_classifier_preserves_confirmed_and_unverified_paths(self): + path = SCRIPTS / 'notification_events.py' + ns = {'re': re, 'capture_journal_context': lambda **kw: ''} + classify = extract(path, '_classify_pve', 'ProxmoxHookWatcher', ns) + severity_map = extract(path, '_map_severity', 'ProxmoxHookWatcher', ns) + backup_outcome = extract(path, '_backup_outcome', 'ProxmoxHookWatcher', ns) + class Event: + def __init__(self, **kw): self.__dict__.update(kw); self.event_id = 'inert' + class Queue: + def __init__(self): self.items = [] + def put(self, event): self.items.append(event) + ns['NotificationEvent'] = Event + receive = extract(path, 'process_webhook', 'ProxmoxHookWatcher', ns) + class Receiver: + _hostname = 'node-a' + _classify_pve = classify + _map_severity = staticmethod(severity_map) + _backup_outcome = staticmethod(backup_outcome) + def __init__(self): self._queue = Queue() + samples = [('vzdump', 'info', 'Backup', 'job finished'), + ('vzdump', 'warning', 'Backup', 'job incomplete'), + ('', 'warning', 'backup job', 'Backup scheduled')] + for kind, severity, title, message in samples: + with self.subTest(kind=kind, message=message): + event, entity, _ = classify(None, kind, severity, title, message) + self.assertEqual((event, entity), ('backup_complete', 'vm')) + receiver = Receiver() + reply = receive(receiver, {'fields': {'type': kind}, 'severity': severity, + 'title': title, 'message': message}) + self.assertEqual(reply['event_type'], event) + self.assertEqual(len(receiver._queue.items), 1) + emitted = receiver._queue.items[0] + self.assertEqual(emitted.severity, 'WARNING' if severity == 'warning' else 'INFO') + output = self.render(emitted.event_type, emitted.data, 'en') + self.assertIn('Backup outcome unconfirmed', output['title']) + self.assertEqual(output['body'].splitlines()[-1], message) # raw PVE body retained + self.assertEqual(classify(None, 'vzdump', 'error', 'Backup', 'failed')[0], 'backup_fail') + + def test_incomplete_vzdump_log_does_not_certify_a_guest(self): + from typing import Dict, Optional + ns = {'re': re, 'Dict': Dict, 'Optional': Optional, 'Any': Any, + 'runtime_message': lambda key, lang, **kw: key} + parser = extract(SCRIPTS / 'notification_templates.py', '_parse_vzdump_message', namespace=ns) + formatter = extract(SCRIPTS / 'notification_templates.py', '_format_vzdump_body', namespace=ns) + incomplete = parser('INFO: Starting Backup of VM 104 (qemu)') + self.assertEqual(incomplete['vms'][0]['status'], 'unknown') + self.assertNotIn('✅', formatter(incomplete, False, 'en')) + mixed = parser('INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: Starting Backup of VM 105 (lxc)') + self.assertEqual([vm['status'] for vm in mixed['vms']], ['ok', 'unknown']) + formatted = formatter(mixed, False, 'en') + self.assertEqual(formatted.count('✅'), 1) + self.assertNotIn('❌', formatted) + conflicting = parser('INFO: Starting Backup of VM 104 (qemu)\nERROR: backup failed\nINFO: Finished Backup of VM 104 (00:01:00)') + self.assertEqual(conflicting['vms'][0]['status'], 'error') + + def test_backup_render_preserves_success_and_marks_unconfirmed_and_failure(self): + for outcome, message, title_part, body_part in ( + ('confirmed', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'Backup complete', '✅'), + ('unconfirmed', 'INFO: Starting Backup of VM 104 (qemu)', 'Backup outcome unconfirmed', '❔'), + ('failed', 'INFO: Starting Backup of VM 104 (qemu)\nERROR: backup failed', 'Backup error reported', '❌'), + ): + with self.subTest(outcome=outcome): + output = self.render('backup_complete', {'hostname': 'node-a', 'pve_type': 'vzdump', + 'pve_message': message, 'pve_title': 'Backup complete', 'backup_outcome': outcome}, 'en') + self.assertIn(title_part, output['title']) + self.assertIn(body_part, output['body']) + if outcome != 'confirmed': self.assertNotIn('Backup complete', output['title']) + output = self.render('backup_complete', {'hostname': 'node-a', 'vmname': 'vm', 'vmid': '104'}, 'en') + self.assertIn('Backup outcome unconfirmed', output['title']) + self.assertNotIn('successfully', output['body']) + conflict = self.render('backup_complete', {'hostname': 'node-a','backup_outcome':'failed', + 'pve_message':'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nERROR: archive write failed'}, 'en') + self.assertIn('ERROR: archive write failed', conflict['body']) + self.assertNotIn('Backup complete', conflict['title']) + + def test_html_email_badge_and_backup_status_are_context_specific(self): + import html + path = SCRIPTS / 'notification_channels.py' + for lang in ('en', 'de', 'es', 'fr', 'it', 'pt', 'sk', 'sv'): + with self.subTest(lang=lang): + catalog = json.loads((ROOT / 'AppImage/messages' / lang / 'common.json').read_text())['runtime']['notifications'] + def text(key, data=None, **values): + value = catalog['channels'] + for part in key.split('.'): value = value[part] + return value.format(**values) + ns = {'Dict': dict, 'Optional': __import__('typing').Optional, + '_runtime_text': text, '_runtime_notification_text': lambda key, data=None: ''} + build = extract(path, '_build_detail_rows', 'EmailChannel', ns) + fmt = extract(path, '_format_html', 'EmailChannel', ns) + class Email: + _SEV_STYLE = {'OK': {'color':'#16a34a','bg':'#f0fdf4','border':'#bbf7d0'}, + 'CRITICAL': {'color':'#dc2626','bg':'#fef2f2','border':'#fecaca'}, + 'INFO': {'color':'blue','bg':'white','border':'gray'}} + _SEV_DEFAULT = {'color':'#6b7280','bg':'#f9fafb','border':'#e5e7eb'} + subject_prefix = 'ProxMenux' + _build_detail_rows = staticmethod(build) + badge = catalog['channels']['email']['severity']['observation'] + recovery = fmt(Email(), 'No longer reported', 'Body', 'OK', {'_event_type': 'error_resolved', + '_notification_language': lang, '_group': 'health'}) + self.assertIn('>' + badge.upper() + '', recovery) + self.assertIn('color:#6b7280;', recovery) + unrelated = fmt(Email(), 'Reconnected', 'Body', 'OK', {'_event_type': 'node_reconnect', + '_notification_language': lang, '_group': 'cluster'}) + self.assertIn('>' + catalog['channels']['email']['severity']['ok'].upper() + '', unrelated) + for outcome, status in [('confirmed', 'completed'), ('unconfirmed','unconfirmed'), ('failed','failed')]: + email = fmt(Email(), 'Backup', 'Details', 'INFO', {'_event_type': 'backup_complete', + 'backup_outcome': outcome, '_notification_language': lang, '_group': 'backup'}) + self.assertIn(catalog['channels']['email']['status'][status], html.unescape(email)) + badge_label = catalog['channels']['email']['status'][status].upper() + self.assertIn('>' + badge_label + '', html.unescape(email)) + if outcome == 'failed': + self.assertIn('color:#dc2626;font-weight:600;', email) + self.assertIn('background:#fef2f2;', email) + elif outcome == 'unconfirmed': + self.assertIn('background:#f9fafb;', email) + else: + self.assertIn('background:#f0fdf4;', email) + + def test_actual_all_locale_rendering_and_fallback_for_backup_outcomes(self): + import importlib.util + module_path = SCRIPTS / 'notification_templates.py' + spec = importlib.util.spec_from_file_location('isolated_notification_templates', module_path) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) # templates and catalogs only; no manager or sends + samples = { + 'confirmed': 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', + 'unconfirmed': 'INFO: Starting Backup of VM 104 (qemu)', + 'failed': 'INFO: Starting Backup of VM 104 (qemu)\nERROR: backup failed', + } + for lang in ('en', 'de', 'es', 'fr', 'it', 'pt', 'sk', 'sv'): + catalog = json.loads((ROOT / 'AppImage/messages' / lang / 'common.json').read_text())['runtime']['notifications'] + for state, message in samples.items(): + with self.subTest(lang=lang, state=state): + data = {'hostname':'node-with-a-long-name','backup_outcome':state,'pve_type':'vzdump', + 'pve_message':message, 'pve_title':'Backup complete', '_notification_language':lang} + result = module.render_template('backup_complete', data, lang) + if state != 'unconfirmed': + key = 'confirmedTitle' if state == 'confirmed' else 'errorTitle' + self.assertEqual(result['title'], catalog['backup'][key].format(hostname=data['hostname'])) + else: self.assertEqual(result['title'], catalog['templates']['backup_complete']['title'].format(hostname=data['hostname'])) + self.assertNotIn('{hostname}', result['title']) + if state == 'unconfirmed': self.assertIn(catalog['backup']['unconfirmedBody'], result['body']) + if state == 'failed': self.assertIn(catalog['backup']['errorBody'], result['body']) + enriched, _ = module.enrich_with_emojis('backup_complete', result['title'], result['body'], data) + self.assertTrue(enriched.startswith({'confirmed':'💾✅','unconfirmed':'💾❔','failed':'💾❌'}[state])) + recovery = module.render_template('error_resolved', {'hostname':'node','category':'temperature', + 'reason':'Old observation','duration':'3d','original_severity':'WARNING'}, lang) + self.assertEqual(recovery['title'],catalog['templates']['error_resolved']['title'].format(hostname='node',category='temperature',entity_suffix='')) + self.assertNotIn('resolved', recovery['title'].lower()) if lang == 'en' else None + restore = module.render_template('system_restore_completed', {'hostname':'node', 'guests':4, + 'stubs':1,'stale_nodes':2,'components':1,'duration':'2m','warnings_block':'Missing module'},lang) + self.assertIn('Missing module',restore['body']) + self.assertNotIn('fully ready',restore['body'].lower()) + + def test_stale_record_disappearance_is_not_claimed_recovery(self): + data = {'hostname': 'node-a', 'category': 'temperature', 'reason': 'Temperature observation (no longer reported)', + 'original_severity': 'WARNING', 'duration': '2d 0h', 'severity': 'OK'} + output = self.render('error_resolved', data, 'en') + self.assertIn('no longer in active health records', output['body']) + self.assertNotIn('resolved', (output['title'] + output['body']).lower()) + self.assertIn('Time since first observation', output['body']) + + def test_actual_poller_stale_disappearance_keeps_reason_factual(self): + class Store: + def get_active_errors(self): return [] + def is_error_acknowledged(self, key): return False + class Event: + def __init__(self, *args, **kwargs): self.kind, self.severity, self.data = args[:3] + class Queue: + def __init__(self): self.items = [] + def put(self, event): self.items.append(event) + ns = {'time': time, 'json': json, 'NotificationEvent': Event, 'Dict': dict} + poll = extract(SCRIPTS / 'notification_events.py', '_check_persistent_health', 'PollingCollector', ns) + class Collector: + _hostname = 'node-a' + _ENTITY_MAP = {'temperature': ('node', '')} + _first_poll_done = True + _known_errors = {'temp': {'category': 'temperature', 'reason': 'Temperature high', + 'severity': 'WARNING', 'first_seen': '2026-09-25T00:00:00'}} + _notified_severity = {'temp': 'WARNING'} + _last_notified = {'temp': 1} + _queue = Queue() + def _guest_storage_error_is_now_foreign(self, *a): return False + def _save_known_errors_meta(self): pass + with patch.dict(sys.modules, {'health_persistence': types.SimpleNamespace(health_persistence=Store())}): + poll(Collector()) + events = Collector._queue.items + self.assertEqual(len(events), 1) + self.assertEqual((events[0].kind, events[0].severity), ('error_resolved', 'OK')) + self.assertEqual(events[0].data['reason'], 'Temperature high (no longer reported)') + rendered = self.render(events[0].kind, events[0].data, 'en') + self.assertNotIn('recovered', rendered['body'].lower()) + + def test_warning_and_clean_restore_keep_only_reported_outcome(self): + for warnings in ('', '⚠️ Boot sanity: missing modules\n'): + with self.subTest(warnings=warnings): + events = [] + ns = {'request': types.SimpleNamespace(remote_addr='127.0.0.1', get_json=lambda **kw: { + 'hostname': 'node-a', 'guests': '2', 'stubs': '0', 'stale_nodes': '0', + 'components': 'none', 'duration': '2m', + 'warnings': 'missing modules' if warnings else ''}), + 'notification_manager': types.SimpleNamespace(emit_event=lambda **kw: events.append(kw)), + 'jsonify': lambda obj: obj} + handler = extract(SCRIPTS / 'flask_notification_routes.py', 'internal_restore_event', namespace=ns) + response, status = handler() + self.assertEqual((status, response['event_type']), (200, 'system_restore_completed')) + event = events[0] + self.assertEqual(event['severity'], 'WARNING' if warnings else 'INFO') + result = self.render(event['event_type'], event['data'], 'en') + self.assertIn('Post-restore tasks completed', result['body']) + self.assertNotIn('fully ready', result['body']) + if warnings: self.assertIn('missing modules', result['body']) + + def test_missing_key_fallback_and_synthetic_translation(self): + catalog = copy.deepcopy(self.catalog) + translated = copy.deepcopy(self.catalog) + for event, fields in EXPECTED.items(): + for field in fields: translated['runtime']['notifications']['templates'][event].pop(field) + _, render = renderer(catalog, translated) + for event, fields in EXPECTED.items(): + for field in fields: + if field not in ('title', 'body'): + continue + self.assertEqual(render(event, {'category': 'disk'}, 'it')[field], + render(event, {'category': 'disk'}, 'en')[field]) + translated['runtime']['notifications']['templates']['error_resolved']['title'] = 'Synthetic observation: {category}' + _, render = renderer(catalog, translated) + self.assertEqual(render('error_resolved', {'category': 'disk'}, 'it')['title'], 'Synthetic observation: disk') + + def test_rich_backup_icon_tracks_outcome_and_digest_default_is_neutral(self): + tree = ast.parse((SCRIPTS / 'notification_templates.py').read_text()) + def assign(name): + return ast.literal_eval(next(n.value for n in tree.body if isinstance(n, ast.Assign) + and any(isinstance(t, ast.Name) and t.id == name for t in n.targets))) + icons = assign('EVENT_EMOJI') + self.assertNotIn('✅', icons['backup_complete']) + ns = {'TEMPLATES': assign('TEMPLATES'), 'EVENT_EMOJI': icons, + 'Dict': dict, 'Any': Any, + 'CATEGORY_EMOJI': assign('CATEGORY_EMOJI'), 'SEVERITY_ICONS': assign('SEVERITY_ICONS'), + 'FIELD_EMOJI': assign('FIELD_EMOJI'), '_localized_template_labels': lambda *a: {}, + '_lxc_update_label_icons': lambda *a: {}} + enrich = extract(SCRIPTS / 'notification_templates.py', 'enrich_with_emojis', namespace=ns) + for state, icon in [('confirmed', '💾✅'), ('unconfirmed', '💾❔'), ('failed', '💾❌')]: + with self.subTest(state=state): + title, body = enrich('backup_complete', 'node-a: Backup', 'Backup report', + {'backup_outcome': state, '_notification_language': 'en'}) + self.assertTrue(title.startswith(icon), title) + + def test_uncertain_outcomes_have_no_blanket_success_icon(self): + tree = ast.parse((SCRIPTS / 'notification_templates.py').read_text()) + icon_map = ast.literal_eval(next(n.value for n in tree.body if isinstance(n, ast.Assign) + and any(isinstance(t, ast.Name) and t.id == 'EVENT_EMOJI' for t in n.targets))) + for event in ('error_resolved', 'system_restore_completed'): + self.assertNotIn('✅', icon_map[event], event) + self.assertNotIn('✅', icon_map['backup_complete']) # buffered digest has no outcome metadata + + +if __name__ == '__main__': unittest.main() diff --git a/AppImage/messages/de/common.json b/AppImage/messages/de/common.json index 59447cba..0cb5b7c0 100644 --- a/AppImage/messages/de/common.json +++ b/AppImage/messages/de/common.json @@ -6266,8 +6266,8 @@ "label": "Neues Gesundheitsproblem" }, "error_resolved": { - "title": "{hostname}: Gelöst – {category}{entity_suffix}", - "body": "Das Problem {category} wurde behoben.\n{reason}\n🚦 Vorheriger Schweregrad: {original_severity}\n⏱️ Dauer: {duration}", + "title": "{hostname}: Nicht mehr gemeldet – {category}{entity_suffix}", + "body": "Das Problem in der Kategorie {category} ist nicht mehr in den aktiven Zustandsmeldungen enthalten.\n{reason}\n🚦 Vorheriger Schweregrad: {original_severity}\n⏱️ Zeit seit der ersten Meldung: {duration}", "label": "Wiederherstellungsbenachrichtigung" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Sicherung gestartet" }, "backup_complete": { - "title": "{hostname} → {storage}: Sicherung abgeschlossen – {vmname} ({vmid})", - "body": "Die Sicherung von {vmname} (ID: {vmid}) wurde am {storage} erfolgreich abgeschlossen.\nGröße: {size}", + "title": "{hostname}: Backup-Ergebnis nicht bestätigt", + "body": "Das Backup-Ergebnis lässt sich anhand dieser Meldung nicht bestätigen.", "label": "Sicherung abgeschlossen" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: Host-Wiederherstellung abgeschlossen", - "body": "Aufgaben nach der Wiederherstellung im Hintergrund ausgeführt.\n\nGäste haben sich beworben: {guests}\nBind-Mount-Stubs: {stubs}\nVeraltete Knotenverzeichnisse entfernt: {stale_nodes}\nKomponenten neu installiert: {components}\nDauer: {duration}\n{warnings_block}\nDer Knoten ist nun vollständig einsatzbereit.", + "body": "Aufgaben nach der Wiederherstellung im Hintergrund abgeschlossen.\n\nÜbernommene Gäste: {guests}\nBind-Mount-Platzhalter: {stubs}\nEntfernte veraltete Knotenverzeichnisse: {stale_nodes}\nNeu installierte Komponenten: {components}\nDauer: {duration}\n{warnings_block}", "label": "Host-Wiederherstellung abgeschlossen" }, "system_problem": { @@ -6910,7 +6910,8 @@ "warning": "Warnung", "info": "Informationen", "ok": "Gelöst", - "default": "Hinweis" + "default": "Hinweis", + "observation": "Nicht mehr gemeldet" }, "groups": { "vm_ct": "Virtuelle Maschine / Container", @@ -6978,7 +6979,8 @@ "status": { "failed": "Fehlgeschlagen", "completed": "Abgeschlossen", - "started": "Gestartet" + "started": "Gestartet", + "unconfirmed": "Nicht bestätigt" }, "report": "{group} Bericht", "details": "Details", @@ -6988,6 +6990,13 @@ }, "temperature": { "sampleSpan": "Die hohen Messwerte erstrecken sich über {duration}." + }, + "backup": { + "confirmedTitle": "{hostname}: Backup abgeschlossen", + "confirmedBody": "Backup erfolgreich abgeschlossen.", + "errorTitle": "{hostname}: Backup-Fehler gemeldet", + "errorBody": "Der Backup-Bericht enthält einen Fehler.", + "unconfirmedBody": "Das Backup-Ergebnis ist nicht bestätigt." } } } diff --git a/AppImage/messages/en/common.json b/AppImage/messages/en/common.json index cc497d8f..abd42a64 100644 --- a/AppImage/messages/en/common.json +++ b/AppImage/messages/en/common.json @@ -6251,5 +6251,5 @@ "cancel": "Cancel" } }, - "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"The {category} issue has been resolved.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Duration: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname} → {storage}: Backup complete — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}\nThe node is now fully ready to use.","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or run the post-install menu (option 2) → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Settings > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started"}}},"temperature":{"sampleSpan":"High samples span {duration}."}}} + "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: No longer reported - {category}{entity_suffix}","body":"The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname}: Backup outcome unconfirmed","body":"The backup outcome could not be confirmed from this notice.","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or run the post-install menu (option 2) → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Settings > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice","observation":"No longer reported"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"backup":{"confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed."}}} } diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index dc7b0198..86786662 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -6266,8 +6266,8 @@ "label": "Nuevo problema de salud" }, "error_resolved": { - "title": "{hostname}: Resuelto - {category}{entity_suffix}", - "body": "El problema {category} se ha resuelto.\n{reason}\n🚦 Gravedad anterior: {original_severity}\n⏱️ Duración: {duration}", + "title": "{hostname}: Ya no se informa de {category}{entity_suffix}", + "body": "El problema de {category} ya no figura entre las incidencias de salud activas.\n{reason}\n🚦 Gravedad anterior: {original_severity}\n⏱️ Tiempo desde la primera observación: {duration}", "label": "Notificación de recuperación" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Backup iniciado" }, "backup_complete": { - "title": "{hostname} → {storage}: Backup completado — {vmname} ({vmid})", - "body": "El backup de {vmname} (ID: {vmid}) se ha completado correctamente en {storage}.\nTamaño: {size}", + "title": "{hostname}: resultado del backup sin confirmar", + "body": "Esta notificación no permite confirmar el resultado del backup.", "label": "Backup completado" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: restauración del host finalizada", - "body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nGuests aplicados: {guests}\nStubs de bind mount: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}\nEl nodo está listo para usarse.", + "body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nConfiguraciones de invitados aplicadas: {guests}\nDirectorios auxiliares de montajes bind: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}", "label": "Restauración del host completada" }, "system_problem": { @@ -6910,7 +6910,8 @@ "warning": "Advertencia", "info": "Información", "ok": "Resuelto", - "default": "Aviso" + "default": "Aviso", + "observation": "Ya no se informa" }, "groups": { "vm_ct": "Máquina virtual / Contenedor", @@ -6978,7 +6979,8 @@ "status": { "failed": "Fallido", "completed": "Completado", - "started": "Iniciado" + "started": "Iniciado", + "unconfirmed": "Sin confirmar" }, "report": "Informe de {group}", "details": "Detalles", @@ -6988,6 +6990,13 @@ }, "temperature": { "sampleSpan": "Lecturas altas registradas a lo largo de {duration}." + }, + "backup": { + "confirmedTitle": "{hostname}: backup completado", + "confirmedBody": "Backup completado correctamente.", + "errorTitle": "{hostname}: error notificado en el backup", + "errorBody": "El informe del backup contiene un error.", + "unconfirmedBody": "El resultado del backup no está confirmado." } } } diff --git a/AppImage/messages/fr/common.json b/AppImage/messages/fr/common.json index 81b5efe3..d3ebeea9 100644 --- a/AppImage/messages/fr/common.json +++ b/AppImage/messages/fr/common.json @@ -6266,8 +6266,8 @@ "label": "Nouveau problème de santé" }, "error_resolved": { - "title": "{hostname} : Résolu - {category}{entity_suffix}", - "body": "Le problème {category} a été résolu.\n{reason}\n🚦 Gravité précédente : {original_severity}\n⏱️ Durée : {duration}", + "title": "{hostname} : Plus signalé – {category}{entity_suffix}", + "body": "Le problème {category} ne figure plus parmi les alertes de santé actives.\n{reason}\n🚦 Gravité précédente : {original_severity}\n⏱️ Temps depuis la première observation : {duration}", "label": "Notification de récupération" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Sauvegarde démarrée" }, "backup_complete": { - "title": "{hostname} → {storage} : Sauvegarde terminée — {vmname} ({vmid})", - "body": "La sauvegarde de {vmname} (ID : {vmid}) s'est terminée avec succès le {storage}.\nTaille : {size}", + "title": "{hostname} : résultat de la sauvegarde non confirmé", + "body": "Cette notification ne permet pas de confirmer le résultat de la sauvegarde.", "label": "Sauvegarde terminée" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname} : restauration de l'hôte terminée", - "body": "Tâches post-restauration effectuées en arrière-plan.\n\nInvités postulés : {guests}\nTalons de montage liés : {stubs}\nRépertoires de nœuds obsolètes supprimés : {stale_nodes}\nComposants réinstallés : {components}\nDurée : {duration}\n{warnings_block}\nLe nœud est maintenant entièrement prêt à être utilisé.", + "body": "Tâches après restauration terminées en arrière-plan.\n\nInvités appliqués : {guests}\nRépertoires de support des montages bind : {stubs}\nRépertoires de nœuds obsolètes supprimés : {stale_nodes}\nComposants réinstallés : {components}\nDurée : {duration}\n{warnings_block}", "label": "Restauration de l'hôte terminée" }, "system_problem": { @@ -6910,7 +6910,8 @@ "warning": "Avertissement", "info": "Informations", "ok": "Résolu", - "default": "Avis" + "default": "Avis", + "observation": "Plus signalé" }, "groups": { "vm_ct": "Machine Virtuelle / Conteneur", @@ -6978,7 +6979,8 @@ "status": { "failed": "Échec", "completed": "Terminé", - "started": "Commencé" + "started": "Commencé", + "unconfirmed": "Non confirmé" }, "report": "Rapport {group}", "details": "Détails", @@ -6988,6 +6990,13 @@ }, "temperature": { "sampleSpan": "Les relevés élevés s'étendent sur {duration}." + }, + "backup": { + "confirmedTitle": "{hostname} : sauvegarde terminée", + "confirmedBody": "Sauvegarde terminée avec succès.", + "errorTitle": "{hostname} : erreur signalée lors de la sauvegarde", + "errorBody": "Le rapport de sauvegarde contient une erreur.", + "unconfirmedBody": "Le résultat de la sauvegarde n’est pas confirmé." } } } diff --git a/AppImage/messages/it/common.json b/AppImage/messages/it/common.json index b8e733e8..2b7f0990 100644 --- a/AppImage/messages/it/common.json +++ b/AppImage/messages/it/common.json @@ -6266,8 +6266,8 @@ "label": "Nuovo problema sanitario" }, "error_resolved": { - "title": "{hostname}: risolto - {category}{entity_suffix}", - "body": "Il problema {category} è stato risolto.\n{reason}\n🚦 Gravità precedente: {original_severity}\n⏱️ Durata: {duration}", + "title": "{hostname}: segnalazione non più attiva - {category}{entity_suffix}", + "body": "Il problema relativo a {category} non figura più tra le segnalazioni di salute attive.\n{reason}\n🚦 Gravità precedente: {original_severity}\n⏱️ Tempo dalla prima segnalazione: {duration}", "label": "Notifica di recupero" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Backup avviato" }, "backup_complete": { - "title": "{hostname} → {storage}: Backup completato — {vmname} ({vmid})", - "body": "Backup di {vmname} (ID: {vmid}) completato con successo su {storage}.\nTaglia: {size}", + "title": "{hostname}: esito del backup non confermato", + "body": "Questa notifica non consente di confermare l’esito del backup.", "label": "Backup completato" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: ripristino dell'host terminato", - "body": "Attività post-ripristino completate in background.\n\nGli ospiti hanno presentato domanda: {guests}\nStub con montaggio tramite collegamento: {stubs}\nDirectory dei nodi obsolete rimosse: {stale_nodes}\nComponenti reinstallati: {components}\nDurata: {duration}\n{warnings_block}\nIl nodo è ora completamente pronto per l'uso.", + "body": "Attività post-ripristino completate in background.\n\nConfigurazioni guest copiate: {guests}\nDirectory di supporto per montaggi bind create: {stubs}\nDirectory obsolete dei nodi rimosse: {stale_nodes}\nComponenti reinstallati: {components}\nDurata: {duration}\n{warnings_block}", "label": "Ripristino dell'host completato" }, "system_problem": { @@ -6910,7 +6910,8 @@ "warning": "Avvertimento", "info": "Informazioni", "ok": "Risolto", - "default": "Avviso" + "default": "Avviso", + "observation": "Non più segnalato" }, "groups": { "vm_ct": "Macchina virtuale/Contenitore", @@ -6978,7 +6979,8 @@ "status": { "failed": "Fallito", "completed": "Completato", - "started": "Iniziato" + "started": "Iniziato", + "unconfirmed": "Non confermato" }, "report": "{group} Rapporto", "details": "Dettagli", @@ -6988,6 +6990,13 @@ }, "temperature": { "sampleSpan": "Intervallo dei campioni sopra soglia: {duration}." + }, + "backup": { + "confirmedTitle": "{hostname}: backup completato", + "confirmedBody": "Backup completato correttamente.", + "errorTitle": "{hostname}: errore segnalato nel backup", + "errorBody": "Il rapporto del backup contiene un errore.", + "unconfirmedBody": "L’esito del backup non è confermato." } } } diff --git a/AppImage/messages/pt/common.json b/AppImage/messages/pt/common.json index 0527db09..98580f03 100644 --- a/AppImage/messages/pt/common.json +++ b/AppImage/messages/pt/common.json @@ -6266,8 +6266,8 @@ "label": "Novo problema de saúde" }, "error_resolved": { - "title": "{hostname}: Resolvido - {category}{entity_suffix}", - "body": "O problema {category} foi resolvido.\n{reason}\n🚦 Gravidade anterior: {original_severity}\n⏱️ Duração: {duration}", + "title": "{hostname}: Já não comunicado – {category}{entity_suffix}", + "body": "O problema de {category} já não consta dos registos de saúde ativos.\n{reason}\n🚦 Gravidade anterior: {original_severity}\n⏱️ Tempo desde a primeira observação: {duration}", "label": "Notificação de recuperação" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Backup iniciado" }, "backup_complete": { - "title": "{hostname} → {storage}: Backup concluído — {vmname} ({vmid})", - "body": "Backup de {vmname} (ID: {vmid}) concluído com sucesso em {storage}.\nTamanho: {size}", + "title": "{hostname}: resultado do backup não confirmado", + "body": "Esta notificação não permite confirmar o resultado do backup.", "label": "Backup concluído" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: restauração do host concluída", - "body": "Tarefas pós-restauração concluídas em segundo plano.\n\nConvidados inscritos: {guests}\nStubs de montagem de ligação: {stubs}\nDiretórios de nó obsoletos removidos: {stale_nodes}\nComponentes reinstalados: {components}\nDuração: {duration}\n{warnings_block}\nO nó agora está totalmente pronto para uso.", + "body": "Tarefas pós-restauro concluídas em segundo plano.\n\nConvidados aplicados: {guests}\nDiretórios auxiliares de montagens bind: {stubs}\nDiretórios de nós obsoletos removidos: {stale_nodes}\nComponentes reinstalados: {components}\nDuração: {duration}\n{warnings_block}", "label": "Restauração do host concluída" }, "system_problem": { @@ -6910,7 +6910,8 @@ "warning": "Aviso", "info": "Informação", "ok": "Resolvido", - "default": "Aviso" + "default": "Aviso", + "observation": "Já não comunicado" }, "groups": { "vm_ct": "Máquina Virtual/Contêiner", @@ -6978,7 +6979,8 @@ "status": { "failed": "Falhou", "completed": "Concluído", - "started": "Iniciado" + "started": "Iniciado", + "unconfirmed": "Não confirmado" }, "report": "Relatório {group}", "details": "Detalhes", @@ -6988,6 +6990,13 @@ }, "temperature": { "sampleSpan": "As amostras elevadas abrangem {duration}." + }, + "backup": { + "confirmedTitle": "{hostname}: backup concluído", + "confirmedBody": "Backup concluído com sucesso.", + "errorTitle": "{hostname}: erro comunicado no backup", + "errorBody": "O relatório do backup contém um erro.", + "unconfirmedBody": "O resultado do backup não está confirmado." } } } diff --git a/AppImage/messages/sv/common.json b/AppImage/messages/sv/common.json index a61454c0..405e1553 100644 --- a/AppImage/messages/sv/common.json +++ b/AppImage/messages/sv/common.json @@ -6266,8 +6266,8 @@ "label": "Nytt hälsoproblem" }, "error_resolved": { - "title": "{hostname}: Löst - {category}{entity_suffix}", - "body": "{category}-problemet har lösts.\n{reason}\n🚦 Tidigare svårighetsgrad: {original_severity}\n⏱️ Varaktighet: {duration}", + "title": "{hostname}: Rapporteras inte längre – {category}{entity_suffix}", + "body": "Problemet i kategorin {category} finns inte längre bland aktiva hälsoposter.\n{reason}\n🚦 Tidigare allvarlighetsgrad: {original_severity}\n⏱️ Tid sedan första observationen: {duration}", "label": "Återställningsmeddelande" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Säkerhetskopiering startade" }, "backup_complete": { - "title": "{hostname} → {storage}: Säkerhetskopiering klar — {vmname} ({vmid})", - "body": "Säkerhetskopiering av {vmname} (ID: {vmid}) slutfördes framgångsrikt på {storage}.\nStorlek: {size}", + "title": "{hostname}: säkerhetskopians resultat obekräftat", + "body": "Det går inte att bekräfta säkerhetskopians resultat utifrån denna avisering.", "label": "Säkerhetskopieringen är klar" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: Värdåterställning avslutad", - "body": "Uppgifter efter återställning slutförda i bakgrunden.\n\nGäster ansökte: {guests}\nBind-monterade stubbar: {stubs}\nInaktuella nodkataloger har tagits bort: {stale_nodes}\nKomponenter installerade om: {components}\nVaraktighet: {duration}\n{warnings_block}\nNoden är nu helt redo att användas.", + "body": "Åtgärder efter återställning slutfördes i bakgrunden.\n\nGäster tillämpade: {guests}\nHjälpkataloger för bind-monteringar: {stubs}\nFöråldrade nodkataloger borttagna: {stale_nodes}\nKomponenter ominstallerade: {components}\nVaraktighet: {duration}\n{warnings_block}", "label": "Värdåterställning slutförd" }, "system_problem": { @@ -6910,7 +6910,8 @@ "warning": "Varning", "info": "Information", "ok": "Löst", - "default": "Observera" + "default": "Observera", + "observation": "Rapporteras inte längre" }, "groups": { "vm_ct": "Virtuell maskin / behållare", @@ -6978,7 +6979,8 @@ "status": { "failed": "Misslyckades", "completed": "Klar", - "started": "Startat" + "started": "Startat", + "unconfirmed": "Obekräftat" }, "report": "{group} Rapportera", "details": "Detaljer", @@ -6988,6 +6990,13 @@ }, "temperature": { "sampleSpan": "De höga mätvärdena sträcker sig över {duration}." + }, + "backup": { + "confirmedTitle": "{hostname}: säkerhetskopiering klar", + "confirmedBody": "Säkerhetskopieringen slutfördes utan fel.", + "errorTitle": "{hostname}: fel rapporterat vid säkerhetskopiering", + "errorBody": "Rapporten om säkerhetskopieringen innehåller ett fel.", + "unconfirmedBody": "Säkerhetskopieringens resultat är inte bekräftat." } } } diff --git a/AppImage/scripts/notification_channels.py b/AppImage/scripts/notification_channels.py index 60e0afd4..6580826f 100644 --- a/AppImage/scripts/notification_channels.py +++ b/AppImage/scripts/notification_channels.py @@ -1037,6 +1037,21 @@ class EmailChannel(NotificationChannel): # Determine group for section header event_type = data.get('_event_type', '') + if event_type == 'error_resolved': + sev.update(self._SEV_DEFAULT) + sev['label'] = _runtime_text('email.severity.observation', data) + elif event_type == 'backup_complete': + outcome = data.get('backup_outcome') + if outcome == 'confirmed': + sev.update(self._SEV_STYLE['OK']) + status = 'completed' + elif outcome == 'failed': + sev.update(self._SEV_STYLE['CRITICAL']) + status = 'failed' + else: + sev.update(self._SEV_DEFAULT) + status = 'unconfirmed' + sev['label'] = _runtime_text(f'email.status.{status}', data) group = data.get('_group', 'other') # Keep unbroken recorded text inside the temperature email's table. # Both properties are inline for mail clients; other events retain @@ -1221,7 +1236,9 @@ class EmailChannel(NotificationChannel): v = str(value).strip() if value else '' if not v or v == '0' and original_label not in ('Failures',): return - if fmt == 'severity': + if fmt == 'backup_error': + rows.append((esc(label), f'{esc(v)}')) + elif fmt == 'severity': sev_colors = { 'CRITICAL': '#dc2626', 'WARNING': '#d97706', 'INFO': '#2563eb', 'OK': '#16a34a', @@ -1254,9 +1271,13 @@ class EmailChannel(NotificationChannel): # tell which target the backup ran against. Reported gap: emails # showed no way to distinguish which PBS failed with 2+ configured. _add('Storage', data.get('storage') or data.get('storage_name'), 'code') - status_key = 'failed' if 'fail' in event_type else 'completed' if 'complete' in event_type else 'started' + if event_type == 'backup_complete' and data.get('backup_outcome') != 'confirmed': + status_key = ('failed' if data.get('backup_outcome') == 'failed' + else 'unconfirmed') + else: + status_key = 'failed' if 'fail' in event_type else 'completed' if 'complete' in event_type else 'started' _add('Status', _runtime_text(f'email.status.{status_key}', language_data), - 'severity' if 'fail' in event_type else '') + 'backup_error' if status_key == 'failed' else '') _add('Size', data.get('size')) _add('Duration', data.get('duration')) _add('Snapshot', data.get('snapshot_name'), 'code') diff --git a/AppImage/scripts/notification_events.py b/AppImage/scripts/notification_events.py index 6269f6ad..c3eded8d 100644 --- a/AppImage/scripts/notification_events.py +++ b/AppImage/scripts/notification_events.py @@ -3254,7 +3254,7 @@ class PollingCollector: reason_lines = (reason or '').split('\n') reason_summary = reason_lines[0] if reason_lines else '' - # Try to extract device info for a clean "Device: xxx (recovered)" line + # Keep the earlier device context without asserting recovery. device_line = '' for line in reason_lines: if 'Device:' in line or 'Device not currently' in line or '/dev/' in line: @@ -3267,11 +3267,11 @@ class PollingCollector: break if reason_summary and device_line: - clean_reason = f'{reason_summary}\n{device_line} (recovered)' + clean_reason = f'{reason_summary}\n{device_line} (no longer reported)' elif reason_summary: - clean_reason = f'{reason_summary} (recovered)' + clean_reason = f'{reason_summary} (no longer reported)' else: - clean_reason = 'Condition resolved' + clean_reason = 'Condition no longer reported' # `original_severity` must match what the user actually saw # in the most-recent notification for this error, not the @@ -4289,6 +4289,51 @@ class ProxmoxHookWatcher: def _hostname(self) -> str: return _hostname() + @staticmethod + def _backup_outcome(severity: str, message: str) -> str: + """Distinguish explicit failure, complete guest logs and unknown results.""" + text = str(message or '') + if severity in ('error', 'err', 'critical') or re.search( + r'(?im)^\s*(?:ERROR:|TASK ERROR:|.*\bStatus\s+ERROR\b)', text): + return 'failed' + if severity not in ('info', 'ok', 'success') or re.search( + r'(?im)(?:^\s*WARNING:|\bWARNINGS\s*:\s*\d+)', text): + return 'unconfirmed' + starts = re.findall(r'(?im)\bStarting Backup of VM (\d+)\s*\(', text) + finished = re.findall(r'(?im)\bFinished Backup of VM (\d+)\s*\(', text) + lines = text.splitlines() + table_outcome = None + for index, header in enumerate(lines): + if not re.match(r'\s*VMID\s+Name\s+Status\b', header, re.IGNORECASE): + continue + status_start = header.find('Status') + status_end = header.find('Time', status_start) + if status_start < 0 or status_end < 0: + break + rows = [] + for line in lines[index + 1:]: + if re.match(r'\s*Total\b', line, re.IGNORECASE): + table_outcome = ('confirmed' if rows and all(status == 'OK' for status in rows) + else 'unconfirmed') + break + if not line.strip(): + break + if not re.match(r'\s*\d+\s+', line): + break + status = line[status_start:status_end].strip().upper() + if status == 'ERROR': + return 'failed' + rows.append(status) + break + if table_outcome == 'unconfirmed': + return 'unconfirmed' + if starts: + return 'confirmed' if sorted(starts) == sorted(finished) else 'unconfirmed' + if table_outcome == 'confirmed' or re.search( + r'(?im)^\s*(?:INFO:\s*)?TASK OK\s*$', text): + return 'confirmed' + return 'unconfirmed' + def process_webhook(self, payload: dict) -> dict: """Process an incoming Proxmox webhook payload. @@ -4347,6 +4392,13 @@ class ProxmoxHookWatcher: 'title': title or event_type, 'job_id': pve_job_id, } + if event_type in ('backup_complete', 'backup_fail'): + # This is presentation metadata, not a new event/toggle/delivery path. + data['backup_outcome'] = ( + 'failed' if event_type == 'backup_fail' else + self._backup_outcome(severity_raw, message) if pve_type == 'vzdump' + else 'unconfirmed' + ) if pve_type == 'replication': replication = self._extract_replication_context( diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index b8b20c91..e26018af 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -299,7 +299,7 @@ def _parse_vzdump_message(message: str) -> Optional[Dict[str, Any]]: current_vm = { 'vmid': m_start.group(1), 'name': '', - 'status': 'ok', + 'status': 'unknown', 'time': '', 'size': '', 'filename': '', @@ -338,15 +338,16 @@ def _parse_vzdump_message(message: str) -> Optional[Dict[str, Any]]: # Finished -> duration m_finish = re.match( r'Finished Backup of VM (\d+)\s+\(([^)]+)\)', clean) - if m_finish: + if m_finish and m_finish.group(1) == current_vm['vmid']: current_vm['time'] = m_finish.group(2) - current_vm['status'] = 'ok' + if current_vm['status'] != 'error': + current_vm['status'] = 'ok' vms.append(current_vm) current_vm = None continue # Error - if clean.startswith('ERROR:') or clean.startswith('TASK ERROR'): + if re.match(r'^\s*(?:ERROR:|TASK ERROR)', line, re.IGNORECASE): if current_vm: current_vm['status'] = 'error' @@ -439,7 +440,7 @@ def _format_vzdump_body(parsed: Dict[str, Any], is_success: bool, for vm in parsed.get('vms', []): status = vm.get('status', '').lower() - icon = '\u2705' if status == 'ok' else '\u274C' + icon = '\u2705' if status == 'ok' else '\u274C' if status == 'error' else '\u2754' # Determine VM/CT type prefix vm_type = vm.get('type', '') @@ -501,7 +502,8 @@ def _format_vzdump_body(parsed: Dict[str, Any], is_success: bool, if vm_count > 0 or parsed.get('total_size'): ok_count = sum(1 for v in parsed.get('vms', []) if v.get('status', '').lower() == 'ok') - fail_count = vm_count - ok_count + fail_count = sum(1 for v in parsed.get('vms', []) + if v.get('status', '').lower() == 'error') summary_parts = [] if vm_count: @@ -779,11 +781,11 @@ TEMPLATES = { # `{entity}` is populated by health_persistence.resolve_error() # (via _entity_from_details) and by PollingCollector's spread of # the original details blob. When absent, _SafeDict elides the - # placeholder and the title collapses back to "Resolved - " + # placeholder and the title collapses back to "No longer reported - " # without a trailing dash. - 'title': '{hostname}: Resolved - {category}{entity_suffix}', - 'body': 'The {category} issue has been resolved.\n{reason}\n\U0001F6A6 Previous severity: {original_severity}\n\u23F1\uFE0F Duration: {duration}', - 'label': 'Recovery notification', + 'title': '{hostname}: No longer reported - {category}{entity_suffix}', + 'body': 'The {category} issue is no longer in active health records.\n{reason}\n\U0001F6A6 Previous severity: {original_severity}\n\u23F1\uFE0F Time since first observation: {duration}', + 'label': 'Health issue no longer reported', 'group': 'health', 'default_enabled': True, }, @@ -999,9 +1001,9 @@ TEMPLATES = { 'default_enabled': False, }, 'backup_complete': { - 'title': '{hostname} → {storage}: Backup complete — {vmname} ({vmid})', - 'body': 'Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}', - 'label': 'Backup complete', + 'title': '{hostname}: Backup outcome unconfirmed', + 'body': 'The backup outcome could not be confirmed from this notice.', + 'label': 'Backup report', 'group': 'backup', 'default_enabled': True, }, @@ -1270,8 +1272,7 @@ TEMPLATES = { 'Stale node dirs removed: {stale_nodes}\n' 'Components reinstalled: {components}\n' 'Duration: {duration}\n' - '{warnings_block}\n' - 'The node is now fully ready to use.' + '{warnings_block}' ), 'label': 'Host restore completed', 'group': 'services', @@ -1855,6 +1856,16 @@ def render_template(event_type: str, data: Dict[str, Any], ) if localized: template[field] = localized + if event_type == 'backup_complete': + outcome = data.get('backup_outcome') + if outcome == 'confirmed': + template['title'] = runtime_message('backup.confirmedTitle', language, + hostname=data.get('hostname') or _get_hostname()) + template['body'] = runtime_message('backup.confirmedBody', language) + elif outcome == 'failed': + template['title'] = runtime_message('backup.errorTitle', language, + hostname=data.get('hostname') or _get_hostname()) + template['body'] = runtime_message('backup.errorBody', language) # Ensure hostname is always available variables = { @@ -1997,7 +2008,6 @@ def render_template(event_type: str, data: Dict[str, Any], # When the event came from PVE webhook with a full vzdump message, # parse the table/logs and format a rich body instead of the sparse template. pve_message = data.get('pve_message', '') - pve_title = data.get('pve_title', '') # Check for custom formatter function formatter_name = template.get('formatter') @@ -2016,13 +2026,18 @@ def render_template(event_type: str, data: Dict[str, Any], if parsed: is_success = (event_type == 'backup_complete') body_text = _format_vzdump_body(parsed, is_success, language=language) - # Preserve PVE's source title for English, but never leak it into a - # deterministic localized notification. - if pve_title and requested_language == 'en': - title = pve_title + if event_type == 'backup_complete' and data.get('backup_outcome') == 'failed': + error_lines = [line.strip() for line in pve_message.splitlines() + if re.match(r'^\s*(?:ERROR:|TASK ERROR)', line, re.IGNORECASE)] + if error_lines: + body_text += '\n' + '\n'.join(error_lines) else: # Couldn't parse -- use PVE raw message as body body_text = pve_message.strip() + if event_type == 'backup_complete' and data.get('backup_outcome') != 'confirmed': + key = ('backup.errorBody' if data.get('backup_outcome') == 'failed' + else 'backup.unconfirmedBody') + body_text = runtime_message(key, language) + '\n' + body_text elif event_type == 'system_mail' and pve_message: # System mail -- use PVE message directly (mail bounce, cron, smartd) body_text = pve_message.strip()[:1000] @@ -2166,7 +2181,7 @@ EVENT_EMOJI = { 'host_backup_start': '\U0001F5C4️\U0001F680', # 🗄️🚀 cabinet + rocket 'host_backup_complete': '\U0001F5C4️✅', # 🗄️✅ cabinet + check 'host_backup_fail': '\U0001F5C4️❌', # 🗄️❌ cabinet + cross - 'backup_complete': '\U0001F4BE\u2705', # 💾✅ floppy + check + 'backup_complete': '\U0001F4BE', # 💾 neutral for digests without outcome metadata 'backup_warning': '\U0001F4BE\u26A0\uFE0F', # 💾⚠️ floppy + warning 'backup_fail': '\U0001F4BE\u274C', # 💾❌ floppy + cross 'snapshot_complete': '\U0001F4F8', # camera with flash @@ -2204,14 +2219,14 @@ EVENT_EMOJI = { 'system_startup': '\U0001F680', # rocket (startup) 'system_shutdown': '\u23FB\uFE0F', # power symbol (Unicode) 'system_reboot': '\U0001F504', - 'system_restore_completed': '✅', # check mark + 'system_restore_completed': '\U0001F4CB', # post-restore task report (boot may have warnings) 'system_problem': '\u26A0\uFE0F', 'kernel_warning': '\u26A0\uFE0F', 'service_fail': '\u274C', 'oom_kill': '\U0001F4A3', # bomb # Health 'new_error': '\U0001F198', # SOS - 'error_resolved': '\u2705', + 'error_resolved': '\U0001F4CB', # no longer active in health records, not proven recovery 'error_escalated': '\U0001F53A', # red triangle up 'health_degraded': '\u26A0\uFE0F', 'health_persistent': '\U0001F4CB', # clipboard @@ -2363,6 +2378,10 @@ def enrich_with_emojis(event_type: str, title: str, body: str, severity = data.get('severity', 'INFO') icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '') + if event_type == 'backup_complete': + icon = { + 'confirmed': '💾✅', 'failed': '💾❌', + }.get(str(data.get('backup_outcome') or ''), '💾❔') # Build enriched title: replace severity circle with event-specific icon # Current format: "hostname: Something" -> "ICON hostname: Something" diff --git a/AppImage/scripts/tests/test_notification_runtime_i18n.py b/AppImage/scripts/tests/test_notification_runtime_i18n.py index 14cacb30..f2f79843 100644 --- a/AppImage/scripts/tests/test_notification_runtime_i18n.py +++ b/AppImage/scripts/tests/test_notification_runtime_i18n.py @@ -312,6 +312,7 @@ class RuntimeCatalogTests(unittest.TestCase): "backup_complete", { "hostname": "pve01", "storage": "pbs-main", "vmname": "alpha", "vmid": "100", + "backup_outcome": "confirmed", "pve_title": "Backup job finished", "pve_message": ( "INFO: Starting Backup of VM 100 (qemu)\n" @@ -322,7 +323,7 @@ class RuntimeCatalogTests(unittest.TestCase): }, language="sk", ) - self.assertIn("Záloha dokončená", backup["title"]) + self.assertIn("záloha dokončená", backup["title"]) self.assertNotIn("Backup job finished", backup["title"]) self.assertIn("Veľkosť: 1.5 GiB", backup["body"]) self.assertIn("Trvanie: 00:00:10", backup["body"]) @@ -627,6 +628,7 @@ class RuntimeCatalogTests(unittest.TestCase): "_notification_language": "sk", "_event_type": event_type, "_group": "backup", "hostname": "pve01", "vmid": "100", "vmname": "alpha", "storage": "pbs-main", + "backup_outcome": "confirmed" if event_type == "backup_complete" else "unconfirmed", }, ) self.assertIn(f">{localized_status}<", backup_html) diff --git a/AppImage/scripts/tests/test_vzdump_webhook_truncation.py b/AppImage/scripts/tests/test_vzdump_webhook_truncation.py index a7081a53..ab7e28ee 100644 --- a/AppImage/scripts/tests/test_vzdump_webhook_truncation.py +++ b/AppImage/scripts/tests/test_vzdump_webhook_truncation.py @@ -60,7 +60,7 @@ def _make_long_vzdump_report(): class VzdumpWebhookTruncationTests(unittest.TestCase): - def test_truncating_vzdump_report_at_4096_can_create_false_failed_backup(self): + def test_truncating_vzdump_report_at_4096_leaves_guest_unconfirmed(self): full_message = _make_long_vzdump_report() truncated_message = full_message[:4096] @@ -82,8 +82,8 @@ class VzdumpWebhookTruncationTests(unittest.TestCase): self.assertEqual(truncated_dockflare["name"], "dockflare") self.assertEqual(truncated_dockflare["status"], "") - self.assertIn("❌ dockflare (129)", truncated_body) - self.assertIn("❌ 1 failed", truncated_body) + self.assertIn("❔ dockflare (129)", truncated_body) + self.assertNotIn("❌ 1 failed", truncated_body) def test_webhook_handler_does_not_truncate_message_before_parsing(self): source = (SCRIPTS_DIR / "flask_notification_routes.py").read_text() From 5faf55a746a640ea45862221c50dc70a5df573bf Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Tue, 29 Sep 2026 17:46:37 +0200 Subject: [PATCH 02/14] fix(monitor): reconcile backup outcome reports with maintainer review --- .../tests/test_command_descriptions.py | 12 ++ .../test_notification_outcome_wording.py | 124 ++++++++++++++++-- AppImage/messages/es/common.json | 2 +- AppImage/scripts/notification_events.py | 18 ++- AppImage/scripts/notification_templates.py | 29 +++- .../tests/test_notification_runtime_i18n.py | 19 ++- 6 files changed, 180 insertions(+), 24 deletions(-) diff --git a/.github/scripts/tests/test_command_descriptions.py b/.github/scripts/tests/test_command_descriptions.py index c1e3fd9d..be4ad419 100644 --- a/.github/scripts/tests/test_command_descriptions.py +++ b/.github/scripts/tests/test_command_descriptions.py @@ -130,6 +130,18 @@ class CommandDescriptionsTests(unittest.TestCase): for key in ("temperatureAlertTitle", "temperatureAlertBody", "recordedReason", "recordedDetails"): fallback.setdefault(key, source_fallback[key]) + # Slovak remains the exact upstream catalog; model the + # pending outcome-key generator additions in disposable + # copies rather than modifying its curated values. + if lang == 'sk': + local = temporary['runtime']['notifications'] + source = catalog('en')['runtime']['notifications'] + for key, value in source['backup'].items(): + local.setdefault('backup', {}).setdefault(key, value) + local['channels']['email']['severity'].setdefault( + 'observation', source['channels']['email']['severity']['observation']) + local['channels']['email']['status'].setdefault( + 'unconfirmed', source['channels']['email']['status']['unconfirmed']) path.write_text(json.dumps(temporary, ensure_ascii=False)) # Model steady state after the bot fills these intentional new # messages; keep repository locales and all other leaves intact. diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py index d94b33fa..0bf932dd 100644 --- a/.github/scripts/tests/test_notification_outcome_wording.py +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -18,7 +18,7 @@ EXPECTED = { 'error_resolved': { 'title': '{hostname}: No longer reported - {category}{entity_suffix}', 'body': 'The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}', - 'label': 'Health issue no longer reported', + 'label': 'Recovery notification', }, 'system_restore_completed': { @@ -71,6 +71,47 @@ class OutcomeWording(unittest.TestCase): cls.templates, render = renderer(cls.catalog) cls.render = staticmethod(render) + def test_upstream_slovak_stale_claims_use_english_report_fallback(self): + import importlib.util + spec = importlib.util.spec_from_file_location('isolated_slovak_report', SCRIPTS / 'notification_templates.py') + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + recovered = module.render_template('error_resolved', {'hostname':'node-a', + 'category':'temperature','reason':'old','duration':'3d', + 'original_severity':'WARNING'}, 'sk') + self.assertIn('No longer reported',recovered['title']) + self.assertIn('no longer in active health records',recovered['body']) + restored = module.render_template('system_restore_completed', {'hostname':'node-a', + 'guests':3,'stubs':0,'stale_nodes':0,'components':1,'duration':'2m', + 'warnings_block':'⚠️ Boot check pending'}, 'sk') + self.assertIn('Post-restore tasks completed',restored['body']) + self.assertNotIn('úplne pripravený',restored['body']) + + def test_spanish_restore_names_vms_and_containers_not_invitados(self): + import importlib.util + spec = importlib.util.spec_from_file_location('isolated_spanish_restore', SCRIPTS / 'notification_templates.py') + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + result = module.render_template('system_restore_completed', { + 'hostname':'node-a','guests':3,'stubs':0,'stale_nodes':0, + 'components':1,'duration':'2m','warnings_block':''}, 'es') + self.assertIn('máquinas virtuales y contenedores', result['body']) + self.assertNotIn('invitados', result['body'].lower()) + + def test_settings_labels_stay_at_upstream_values_in_all_locales(self): + import subprocess + for lang in ('en','de','es','fr','it','pt','sk','sv'): + path = f'AppImage/messages/{lang}/common.json' + upstream = json.loads(subprocess.check_output(['git','show',f'eb7cc548:{path}'], cwd=ROOT)) + current = json.loads((ROOT / path).read_text()) + for event in ('backup_complete', 'error_resolved'): + with self.subTest(lang=lang, event=event): + expected = upstream['runtime']['notifications']['templates'][event]['label'] + self.assertEqual(current['runtime']['notifications']['templates'][event]['label'], expected) + if lang == 'en': self.assertEqual(self.templates[event]['label'], expected) + def test_exact_four_english_leaves_match_source_and_catalog(self): for event, fields in EXPECTED.items(): for field, value in fields.items(): @@ -101,7 +142,19 @@ class OutcomeWording(unittest.TestCase): row_ok = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('104','alpha','OK','00:01:00','1.5 GiB','archive') row_warning = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('105','beta','WARNINGS','00:01:00','1.5 GiB','archive') row_error = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('105','beta','ERROR','00:01:00','1.5 GiB','archive') + row_err = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('105','beta','err','00:01:00','1.5 GiB','archive') + truncated = 'INFO: Log output was too long to be displayed. Please see task log for details.' cases = [ + ('vzdump', 'info', header+'\n'+row_ok+'\n'+row_err+'\nTotal running time: 00:02:00', 'failed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\n'+header+'\n'+row_ok+'\nTotal running time: 00:01:00\n'+truncated, 'confirmed'), + ('vzdump', 'info', header+'\n'+row_ok+'\nTotal running time: 00:01:00\n'+truncated, 'confirmed'), + ('vzdump', 'info', header+'\n'+row_err+'\nTotal running time: 00:01:00\n'+truncated, 'failed'), + ('vzdump', 'warning', header+'\n'+row_ok+'\nTotal running time: 00:01:00', 'unconfirmed'), + ('vzdump', 'info', header+'\n'+row_ok, 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\n'+header+'\n'+row_ok, 'unconfirmed'), + ('vzdump', 'warning', header+'\n'+row_err+'\nTotal running time: 00:01:00', 'failed'), + ('vzdump', 'info', header+'\n'+row_ok+'\n'+row_warning+'\nTotal running time: 00:02:00\n'+truncated, 'unconfirmed'), + ('vzdump', 'info', header+'\n'+row_ok+'\nTotal running time: 00:01:00\nERROR: archive write failed', 'failed'), ('vzdump', 'info', header+'\n'+row_ok+'\n'+row_error+'\nTotal running time: 00:02:00', 'failed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\n'+header+'\n'+row_error+'\nTotal running time: 00:02:00', 'failed'), ('vzdump', 'info', header+'\n'+row_error, 'failed'), @@ -175,6 +228,11 @@ class OutcomeWording(unittest.TestCase): parser = extract(SCRIPTS / 'notification_templates.py', '_parse_vzdump_message', namespace=ns) formatter = extract(SCRIPTS / 'notification_templates.py', '_format_vzdump_body', namespace=ns) incomplete = parser('INFO: Starting Backup of VM 104 (qemu)') + table_header = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('VMID','Name','Status','Time','Size','Filename') + table_err = '{:<8}{:<22}{:<10}{:<10}{:<14}{}'.format('104','alpha','err','00:01:00','1.5 GiB','archive') + failed_table = parser(table_header+'\n'+table_err+'\nTotal running time: 00:01:00') + self.assertEqual(failed_table['vms'][0]['status'].lower(), 'error') + self.assertIn('❌', formatter(failed_table, False, 'en')) self.assertEqual(incomplete['vms'][0]['status'], 'unknown') self.assertNotIn('✅', formatter(incomplete, False, 'en')) mixed = parser('INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: Starting Backup of VM 105 (lxc)') @@ -205,16 +263,45 @@ class OutcomeWording(unittest.TestCase): self.assertIn('ERROR: archive write failed', conflict['body']) self.assertNotIn('Backup complete', conflict['title']) + def test_confirmed_title_keeps_single_guest_and_destination_without_misnaming_batches(self): + import importlib.util + spec = importlib.util.spec_from_file_location('isolated_backup_title', SCRIPTS / 'notification_templates.py') + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + actual_render = module.render_template + log = ('INFO: starting new backup job: vzdump 104 --storage PBS-Cloud --mode snapshot\n' + 'INFO: Starting Backup of VM 104 (qemu)\nINFO: VM Name: Alpha\n' + 'INFO: Finished Backup of VM 104 (00:01:00)') + single = actual_render('backup_complete', {'hostname':'node-a','backup_outcome':'confirmed', + 'pve_message':log}, 'en') + self.assertIn('PBS-Cloud',single['title']) + self.assertIn('VM Alpha (104)',single['title']) + batch = actual_render('backup_complete', {'hostname':'node-a','backup_outcome':'confirmed', + 'pve_message':log+'\nINFO: Starting Backup of VM 105 (lxc)\n' + 'INFO: Finished Backup of VM 105 (00:01:00)'}, 'en') + self.assertIn('PBS-Cloud',batch['title']) + self.assertNotIn('Alpha (104)',batch['title']) + no_context = actual_render('backup_complete', {'hostname':'node-a','backup_outcome':'confirmed'}, 'en') + self.assertEqual(no_context['title'], 'node-a: Backup complete') + named = actual_render('backup_complete', {'hostname':'node-a','backup_outcome':'confirmed', + 'pve_message':log.replace('Alpha','Alpha {literal}')}, 'en') + self.assertIn('Alpha {literal} (104)', named['title']) + def test_html_email_badge_and_backup_status_are_context_specific(self): import html path = SCRIPTS / 'notification_channels.py' for lang in ('en', 'de', 'es', 'fr', 'it', 'pt', 'sk', 'sv'): with self.subTest(lang=lang): catalog = json.loads((ROOT / 'AppImage/messages' / lang / 'common.json').read_text())['runtime']['notifications'] + english = self.catalog['runtime']['notifications'] def text(key, data=None, **values): - value = catalog['channels'] - for part in key.split('.'): value = value[part] - return value.format(**values) + def lookup(source): + value = source['channels'] + for part in key.split('.'): + value = value.get(part) if isinstance(value, dict) else None + return value + return (lookup(catalog) or lookup(english) or '').format(**values) ns = {'Dict': dict, 'Optional': __import__('typing').Optional, '_runtime_text': text, '_runtime_notification_text': lambda key, data=None: ''} build = extract(path, '_build_detail_rows', 'EmailChannel', ns) @@ -226,7 +313,7 @@ class OutcomeWording(unittest.TestCase): _SEV_DEFAULT = {'color':'#6b7280','bg':'#f9fafb','border':'#e5e7eb'} subject_prefix = 'ProxMenux' _build_detail_rows = staticmethod(build) - badge = catalog['channels']['email']['severity']['observation'] + badge = catalog['channels']['email']['severity'].get('observation') or english['channels']['email']['severity']['observation'] recovery = fmt(Email(), 'No longer reported', 'Body', 'OK', {'_event_type': 'error_resolved', '_notification_language': lang, '_group': 'health'}) self.assertIn('>' + badge.upper() + '', recovery) @@ -237,8 +324,9 @@ class OutcomeWording(unittest.TestCase): for outcome, status in [('confirmed', 'completed'), ('unconfirmed','unconfirmed'), ('failed','failed')]: email = fmt(Email(), 'Backup', 'Details', 'INFO', {'_event_type': 'backup_complete', 'backup_outcome': outcome, '_notification_language': lang, '_group': 'backup'}) - self.assertIn(catalog['channels']['email']['status'][status], html.unescape(email)) - badge_label = catalog['channels']['email']['status'][status].upper() + label = catalog['channels']['email']['status'].get(status) or english['channels']['email']['status'][status] + self.assertIn(label, html.unescape(email)) + badge_label = label.upper() self.assertIn('>' + badge_label + '', html.unescape(email)) if outcome == 'failed': self.assertIn('color:#dc2626;font-weight:600;', email) @@ -269,16 +357,28 @@ class OutcomeWording(unittest.TestCase): result = module.render_template('backup_complete', data, lang) if state != 'unconfirmed': key = 'confirmedTitle' if state == 'confirmed' else 'errorTitle' - self.assertEqual(result['title'], catalog['backup'][key].format(hostname=data['hostname'])) - else: self.assertEqual(result['title'], catalog['templates']['backup_complete']['title'].format(hostname=data['hostname'])) + expected_title = (catalog.get('backup', {}).get(key) or + self.catalog['runtime']['notifications']['backup'][key]).format(hostname=data['hostname']) + self.assertTrue(result['title'].startswith(expected_title), result['title']) + else: + source = (catalog if catalog.get('backup', {}).get('unconfirmedBody') + else self.catalog['runtime']['notifications']) + self.assertEqual(result['title'], source['templates']['backup_complete']['title'].format(hostname=data['hostname'])) self.assertNotIn('{hostname}', result['title']) - if state == 'unconfirmed': self.assertIn(catalog['backup']['unconfirmedBody'], result['body']) - if state == 'failed': self.assertIn(catalog['backup']['errorBody'], result['body']) + if state == 'unconfirmed': + source = (catalog if catalog.get('backup', {}).get('unconfirmedBody') + else self.catalog['runtime']['notifications']) + self.assertIn(source['backup']['unconfirmedBody'], result['body']) + if state == 'failed': + self.assertIn(catalog.get('backup', {}).get('errorBody') or + self.catalog['runtime']['notifications']['backup']['errorBody'], result['body']) enriched, _ = module.enrich_with_emojis('backup_complete', result['title'], result['body'], data) self.assertTrue(enriched.startswith({'confirmed':'💾✅','unconfirmed':'💾❔','failed':'💾❌'}[state])) recovery = module.render_template('error_resolved', {'hostname':'node','category':'temperature', 'reason':'Old observation','duration':'3d','original_severity':'WARNING'}, lang) - self.assertEqual(recovery['title'],catalog['templates']['error_resolved']['title'].format(hostname='node',category='temperature',entity_suffix='')) + recovery_source = (catalog if catalog.get('backup', {}).get('unconfirmedBody') + else self.catalog['runtime']['notifications']) + self.assertEqual(recovery['title'], recovery_source['templates']['error_resolved']['title'].format(hostname='node',category='temperature',entity_suffix='')) self.assertNotIn('resolved', recovery['title'].lower()) if lang == 'en' else None restore = module.render_template('system_restore_completed', {'hostname':'node', 'guests':4, 'stubs':1,'stale_nodes':2,'components':1,'duration':'2m','warnings_block':'Missing module'},lang) diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index 86786662..43cc5f24 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: restauración del host finalizada", - "body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nConfiguraciones de invitados aplicadas: {guests}\nDirectorios auxiliares de montajes bind: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}", + "body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nConfiguraciones de máquinas virtuales y contenedores aplicadas: {guests}\nDirectorios auxiliares de montajes bind: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}", "label": "Restauración del host completada" }, "system_problem": { diff --git a/AppImage/scripts/notification_events.py b/AppImage/scripts/notification_events.py index c3eded8d..9e949f60 100644 --- a/AppImage/scripts/notification_events.py +++ b/AppImage/scripts/notification_events.py @@ -4296,9 +4296,6 @@ class ProxmoxHookWatcher: if severity in ('error', 'err', 'critical') or re.search( r'(?im)^\s*(?:ERROR:|TASK ERROR:|.*\bStatus\s+ERROR\b)', text): return 'failed' - if severity not in ('info', 'ok', 'success') or re.search( - r'(?im)(?:^\s*WARNING:|\bWARNINGS\s*:\s*\d+)', text): - return 'unconfirmed' starts = re.findall(r'(?im)\bStarting Backup of VM (\d+)\s*\(', text) finished = re.findall(r'(?im)\bFinished Backup of VM (\d+)\s*\(', text) lines = text.splitlines() @@ -4321,15 +4318,24 @@ class ProxmoxHookWatcher: if not re.match(r'\s*\d+\s+', line): break status = line[status_start:status_end].strip().upper() - if status == 'ERROR': + if status in ('ERROR', 'ERR'): return 'failed' rows.append(status) break - if table_outcome == 'unconfirmed': + if severity not in ('info', 'ok', 'success') or re.search( + r'(?im)(?:^\s*WARNING:|\bWARNINGS\s*:\s*\d+)', text): + return 'unconfirmed' + # A present table is authoritative: do not certify an incomplete table + # from a finished guest log, or reject a complete OK table merely + # because the extra diagnostic log was truncated before its finishes. + if table_outcome is not None: + return table_outcome + if any(re.match(r'\s*VMID\s+Name\s+Status\b', line, re.IGNORECASE) + for line in lines): return 'unconfirmed' if starts: return 'confirmed' if sorted(starts) == sorted(finished) else 'unconfirmed' - if table_outcome == 'confirmed' or re.search( + if re.search( r'(?im)^\s*(?:INFO:\s*)?TASK OK\s*$', text): return 'confirmed' return 'unconfirmed' diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index e26018af..5ea8d925 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -252,6 +252,8 @@ def _parse_vzdump_message(message: str) -> Optional[Dict[str, Any]]: vmid = padded[col_starts[0]:col_starts[1]].strip() name = padded[col_starts[1]:col_starts[2]].strip() status = padded[col_starts[2]:col_starts[3]].strip() + if status.lower() in ('err', 'error'): + status = 'error' time_val = padded[col_starts[3]:col_starts[4]].strip() size = padded[col_starts[4]:col_starts[5]].strip() filename = padded[col_starts[5]:].strip() @@ -785,7 +787,7 @@ TEMPLATES = { # without a trailing dash. 'title': '{hostname}: No longer reported - {category}{entity_suffix}', 'body': 'The {category} issue is no longer in active health records.\n{reason}\n\U0001F6A6 Previous severity: {original_severity}\n\u23F1\uFE0F Time since first observation: {duration}', - 'label': 'Health issue no longer reported', + 'label': 'Recovery notification', 'group': 'health', 'default_enabled': True, }, @@ -1003,7 +1005,7 @@ TEMPLATES = { 'backup_complete': { 'title': '{hostname}: Backup outcome unconfirmed', 'body': 'The backup outcome could not be confirmed from this notice.', - 'label': 'Backup report', + 'label': 'Backup complete', 'group': 'backup', 'default_enabled': True, }, @@ -1854,13 +1856,35 @@ def render_template(event_type: str, data: Dict[str, Any], _catalog_value(requested_catalog, key) or _catalog_value(english_catalog, key) ) + # A catalog without the outcome keys predates this report contract. + # Keep its Settings labels, but do not render old recovery, restore + # or backup success claims (e.g. the exact upstream Slovak catalog). + if (event_type in ('backup_complete', 'error_resolved', 'system_restore_completed') + and field in ('title', 'body') + and not _catalog_value(requested_catalog, 'backup.unconfirmedBody')): + localized = _catalog_value(english_catalog, key) if localized: template[field] = localized + backup_title_target = '' if event_type == 'backup_complete': outcome = data.get('backup_outcome') if outcome == 'confirmed': template['title'] = runtime_message('backup.confirmedTitle', language, hostname=data.get('hostname') or _get_hostname()) + parsed_backup = _parse_vzdump_message(str(data.get('pve_message') or '')) + storage = str((parsed_backup or {}).get('storage_name') or data.get('storage') or '').strip() + guests = (parsed_backup or {}).get('vms') or [] + target = [] + if storage: + target.append(storage) + if len(guests) == 1: + guest = guests[0] + kind = 'VM' if guest.get('type') == 'qemu' else 'CT' if guest.get('type') == 'lxc' else 'VM/CT' + name = guest.get('name') or kind + target.append(f"{kind} {name} ({guest['vmid']})" if name != kind + else f"{kind} {guest['vmid']}") + if target: + backup_title_target = ' — ' + ' · '.join(target) template['body'] = runtime_message('backup.confirmedBody', language) elif outcome == 'failed': template['title'] = runtime_message('backup.errorTitle', language, @@ -2003,6 +2027,7 @@ def render_template(event_type: str, data: Dict[str, Any], title = template['title'].format_map(safe_vars) except (ValueError, IndexError): title = template['title'] + title += backup_title_target # ── PVE vzdump special formatting ── # When the event came from PVE webhook with a full vzdump message, diff --git a/AppImage/scripts/tests/test_notification_runtime_i18n.py b/AppImage/scripts/tests/test_notification_runtime_i18n.py index f2f79843..abd1b127 100644 --- a/AppImage/scripts/tests/test_notification_runtime_i18n.py +++ b/AppImage/scripts/tests/test_notification_runtime_i18n.py @@ -50,6 +50,10 @@ class RuntimeCatalogTests(unittest.TestCase): self.assertIsInstance(templates[event_type][field], str) self.assertTrue(templates[event_type][field]) if field in source: + # Upstream Slovak backup title/body still belong to + # the pre-outcome schema; runtime falls back to EN. + if language == 'sk' and event_type == 'backup_complete' and field != 'label': + continue self.assertEqual( _placeholders(templates[event_type][field]), _placeholders(source[field]), @@ -68,10 +72,17 @@ class RuntimeCatalogTests(unittest.TestCase): return result en = flatten(self.catalogs["en"]) + pending_slovak = {"backup.confirmedTitle", "backup.confirmedBody", + "backup.errorTitle", "backup.errorBody", "backup.unconfirmedBody", + "channels.email.severity.observation", "channels.email.status.unconfirmed"} for language, catalog in self.catalogs.items(): translated = flatten(catalog) - self.assertEqual(set(translated), set(en), language) - for key in en: + expected = set(en) - pending_slovak if language == 'sk' else set(en) + self.assertEqual(set(translated), expected, language) + for key in expected: + if language == 'sk' and key in ('templates.backup_complete.title', + 'templates.backup_complete.body'): + continue # exact upstream SK, superseded only at render time self.assertEqual(_placeholders(translated[key]), _placeholders(en[key]), f"{language}:{key}") def test_notification_language_ui_keys_exist_in_both_catalogs(self): @@ -323,7 +334,9 @@ class RuntimeCatalogTests(unittest.TestCase): }, language="sk", ) - self.assertIn("záloha dokončená", backup["title"]) + self.assertIn("Backup complete", backup["title"]) + self.assertIn("pbs-main", backup["title"]) + self.assertIn("VM alpha (100)", backup["title"]) self.assertNotIn("Backup job finished", backup["title"]) self.assertIn("Veľkosť: 1.5 GiB", backup["body"]) self.assertIn("Trvanie: 00:00:10", backup["body"]) From f80e0b678453cd7f1ef5a185a7d60a3ef2fc61d8 Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Tue, 29 Sep 2026 17:58:55 +0200 Subject: [PATCH 03/14] fix(monitor): honor guest terminology and leaf-scoped Slovak fallback --- .../test_notification_outcome_wording.py | 35 +++++++++++++++++-- AppImage/messages/es/common.json | 2 +- AppImage/scripts/notification_templates.py | 18 ++++++---- 3 files changed, 46 insertions(+), 9 deletions(-) diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py index 0bf932dd..f841da7b 100644 --- a/.github/scripts/tests/test_notification_outcome_wording.py +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -88,7 +88,38 @@ class OutcomeWording(unittest.TestCase): self.assertIn('Post-restore tasks completed',restored['body']) self.assertNotIn('úplne pripravený',restored['body']) - def test_spanish_restore_names_vms_and_containers_not_invitados(self): + def test_slovak_fallback_is_per_stale_leaf_not_unrelated_key_presence(self): + import importlib.util + spec = importlib.util.spec_from_file_location('isolated_slovak_future', SCRIPTS / 'notification_templates.py') + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + upstream = json.loads((ROOT / 'AppImage/messages/sk/common.json').read_text())['runtime']['notifications'] + english = self.catalog['runtime']['notifications'] + data = {'hostname': 'node-a', 'category': 'temperature', 'reason': 'old', + 'duration': '3d', 'original_severity': 'WARNING', 'guests': 3} + for event, field in (('error_resolved', 'title'), ('error_resolved', 'body'), + ('system_restore_completed', 'body'), + ('backup_complete', 'title'), ('backup_complete', 'body')): + # A new translation must work independently of another family's key. + translated = copy.deepcopy(upstream) + translated['templates'][event][field] = 'REVIEWED TRANSLATION {hostname}' + with self.subTest(event=event, field=field, case='future translation'): + with patch.object(module, '_load_runtime_catalog', side_effect=lambda lang: translated if lang == 'sk' else english): + result = module.render_template(event, data, 'sk') + self.assertEqual(result[field], 'REVIEWED TRANSLATION node-a') + # Adding outcome keys must not re-enable unrelated stale claims. + stale = copy.deepcopy(upstream) + stale.setdefault('backup', {})['unconfirmedBody'] = 'REVIEWED OUTCOME' + with self.subTest(event=event, field=field, case='stale after key addition'): + with patch.object(module, '_load_runtime_catalog', side_effect=lambda lang: stale if lang == 'sk' else english): + result = module.render_template(event, data, 'sk') + self.assertEqual(result[field], module.render_template(event, data, 'en')[field]) + # The unchanged restore title is not an unsafe readiness claim. + self.assertEqual(module.render_template('system_restore_completed', data, 'sk')['title'], + upstream['templates']['system_restore_completed']['title'].format(**data)) + + def test_spanish_restore_uses_maintainer_guests_terminology(self): import importlib.util spec = importlib.util.spec_from_file_location('isolated_spanish_restore', SCRIPTS / 'notification_templates.py') assert spec is not None and spec.loader is not None @@ -97,7 +128,7 @@ class OutcomeWording(unittest.TestCase): result = module.render_template('system_restore_completed', { 'hostname':'node-a','guests':3,'stubs':0,'stale_nodes':0, 'components':1,'duration':'2m','warnings_block':''}, 'es') - self.assertIn('máquinas virtuales y contenedores', result['body']) + self.assertIn('Configuraciones de guests aplicadas: 3', result['body']) self.assertNotIn('invitados', result['body'].lower()) def test_settings_labels_stay_at_upstream_values_in_all_locales(self): diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index 43cc5f24..e25e0a1a 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: restauración del host finalizada", - "body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nConfiguraciones de máquinas virtuales y contenedores aplicadas: {guests}\nDirectorios auxiliares de montajes bind: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}", + "body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nConfiguraciones de guests aplicadas: {guests}\nDirectorios auxiliares de montajes bind: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}", "label": "Restauración del host completada" }, "system_problem": { diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index 5ea8d925..eba823a9 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -1856,12 +1856,18 @@ def render_template(event_type: str, data: Dict[str, Any], _catalog_value(requested_catalog, key) or _catalog_value(english_catalog, key) ) - # A catalog without the outcome keys predates this report contract. - # Keep its Settings labels, but do not render old recovery, restore - # or backup success claims (e.g. the exact upstream Slovak catalog). - if (event_type in ('backup_complete', 'error_resolved', 'system_restore_completed') - and field in ('title', 'body') - and not _catalog_value(requested_catalog, 'backup.unconfirmedBody')): + # Keep the Slovak catalog with its maintainer. Suppress only the + # exact stale report leaves, not future translations or safe titles. + # An unrelated outcome key cannot version recovery/restore wording. + stale_slovak_reports = { + 'templates.backup_complete.title': '{hostname} → {storage}: Záloha dokončená — {vmname} ({vmid})', + 'templates.backup_complete.body': 'Záloha {vmname} (ID: {vmid}) na úložisku {storage} bola úspešne dokončená.\nVeľkosť: {size}', + 'templates.error_resolved.title': '{hostname}: Vyriešené - {category}{entity_suffix}', + 'templates.error_resolved.body': 'Problém v kategórii {category} bol vyriešený.\n{reason}\n🚦 Predchádzajúca závažnosť: {original_severity}\n⏱️ Trvanie: {duration}', + 'templates.system_restore_completed.body': 'Úlohy po obnove boli dokončené na pozadí.\n\nPoužité VM a LXC: {guests}\nZástupné priečinky bind mountov: {stubs}\nOdstránené zastarané priečinky uzlov: {stale_nodes}\nPreinštalované súčasti: {components}\nTrvanie: {duration}\n{warnings_block}\nUzol je teraz úplne pripravený na použitie.', + } + if (requested_language == 'sk' and key in stale_slovak_reports + and localized == stale_slovak_reports[key]): localized = _catalog_value(english_catalog, key) if localized: template[field] = localized From 2ea261dca3aaee6cbfacb7d72992a3a3426cf17c Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Tue, 29 Sep 2026 18:13:57 +0200 Subject: [PATCH 04/14] test(notifications): keep label checks independent of git history --- .../scripts/tests/test_notification_outcome_wording.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py index f841da7b..824dd016 100644 --- a/.github/scripts/tests/test_notification_outcome_wording.py +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -132,14 +132,14 @@ class OutcomeWording(unittest.TestCase): self.assertNotIn('invitados', result['body'].lower()) def test_settings_labels_stay_at_upstream_values_in_all_locales(self): - import subprocess - for lang in ('en','de','es','fr','it','pt','sk','sv'): + # Frozen from develop eb7cc548; CI shallow checkouts have no base history. + labels = {'en': {'backup_complete': 'Backup complete', 'error_resolved': 'Recovery notification'}, 'de': {'backup_complete': 'Sicherung abgeschlossen', 'error_resolved': 'Wiederherstellungsbenachrichtigung'}, 'es': {'backup_complete': 'Backup completado', 'error_resolved': 'Notificación de recuperación'}, 'fr': {'backup_complete': 'Sauvegarde terminée', 'error_resolved': 'Notification de récupération'}, 'it': {'backup_complete': 'Backup completato', 'error_resolved': 'Notifica di recupero'}, 'pt': {'backup_complete': 'Backup concluído', 'error_resolved': 'Notificação de recuperação'}, 'sk': {'backup_complete': 'Záloha bola dokončená', 'error_resolved': 'Problém bol vyriešený'}, 'sv': {'backup_complete': 'Säkerhetskopieringen är klar', 'error_resolved': 'Återställningsmeddelande'}} + for lang in labels: path = f'AppImage/messages/{lang}/common.json' - upstream = json.loads(subprocess.check_output(['git','show',f'eb7cc548:{path}'], cwd=ROOT)) current = json.loads((ROOT / path).read_text()) for event in ('backup_complete', 'error_resolved'): with self.subTest(lang=lang, event=event): - expected = upstream['runtime']['notifications']['templates'][event]['label'] + expected = labels[lang][event] self.assertEqual(current['runtime']['notifications']['templates'][event]['label'], expected) if lang == 'en': self.assertEqual(self.templates[event]['label'], expected) From e52f52388ad15f627bdd244091894efa9275d479 Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Tue, 29 Sep 2026 22:20:28 +0200 Subject: [PATCH 05/14] fix(monitor): parse PVE 9.2 backup reports and remove Slovak overrides --- .../test_notification_outcome_wording.py | 57 +---- .../scripts/tests/test_notification_pve92.py | 194 ++++++++++++++++++ AppImage/scripts/notification_events.py | 52 ++--- AppImage/scripts/notification_templates.py | 142 ++++++------- 4 files changed, 283 insertions(+), 162 deletions(-) create mode 100644 .github/scripts/tests/test_notification_pve92.py diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py index 824dd016..5b2f8820 100644 --- a/.github/scripts/tests/test_notification_outcome_wording.py +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -59,6 +59,7 @@ def renderer(catalog, translated=None): '_catalog_value': lookup, 'runtime_message': message} from typing import Optional ns['Optional'] = Optional + extract(path, '_parse_vzdump_table', namespace=ns) extract(path, '_parse_vzdump_message', namespace=ns) extract(path, '_format_vzdump_body', namespace=ns) return templates, extract(path, 'render_template', namespace=ns) @@ -71,54 +72,6 @@ class OutcomeWording(unittest.TestCase): cls.templates, render = renderer(cls.catalog) cls.render = staticmethod(render) - def test_upstream_slovak_stale_claims_use_english_report_fallback(self): - import importlib.util - spec = importlib.util.spec_from_file_location('isolated_slovak_report', SCRIPTS / 'notification_templates.py') - assert spec is not None and spec.loader is not None - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - recovered = module.render_template('error_resolved', {'hostname':'node-a', - 'category':'temperature','reason':'old','duration':'3d', - 'original_severity':'WARNING'}, 'sk') - self.assertIn('No longer reported',recovered['title']) - self.assertIn('no longer in active health records',recovered['body']) - restored = module.render_template('system_restore_completed', {'hostname':'node-a', - 'guests':3,'stubs':0,'stale_nodes':0,'components':1,'duration':'2m', - 'warnings_block':'⚠️ Boot check pending'}, 'sk') - self.assertIn('Post-restore tasks completed',restored['body']) - self.assertNotIn('úplne pripravený',restored['body']) - - def test_slovak_fallback_is_per_stale_leaf_not_unrelated_key_presence(self): - import importlib.util - spec = importlib.util.spec_from_file_location('isolated_slovak_future', SCRIPTS / 'notification_templates.py') - assert spec is not None and spec.loader is not None - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - upstream = json.loads((ROOT / 'AppImage/messages/sk/common.json').read_text())['runtime']['notifications'] - english = self.catalog['runtime']['notifications'] - data = {'hostname': 'node-a', 'category': 'temperature', 'reason': 'old', - 'duration': '3d', 'original_severity': 'WARNING', 'guests': 3} - for event, field in (('error_resolved', 'title'), ('error_resolved', 'body'), - ('system_restore_completed', 'body'), - ('backup_complete', 'title'), ('backup_complete', 'body')): - # A new translation must work independently of another family's key. - translated = copy.deepcopy(upstream) - translated['templates'][event][field] = 'REVIEWED TRANSLATION {hostname}' - with self.subTest(event=event, field=field, case='future translation'): - with patch.object(module, '_load_runtime_catalog', side_effect=lambda lang: translated if lang == 'sk' else english): - result = module.render_template(event, data, 'sk') - self.assertEqual(result[field], 'REVIEWED TRANSLATION node-a') - # Adding outcome keys must not re-enable unrelated stale claims. - stale = copy.deepcopy(upstream) - stale.setdefault('backup', {})['unconfirmedBody'] = 'REVIEWED OUTCOME' - with self.subTest(event=event, field=field, case='stale after key addition'): - with patch.object(module, '_load_runtime_catalog', side_effect=lambda lang: stale if lang == 'sk' else english): - result = module.render_template(event, data, 'sk') - self.assertEqual(result[field], module.render_template(event, data, 'en')[field]) - # The unchanged restore title is not an unsafe readiness claim. - self.assertEqual(module.render_template('system_restore_completed', data, 'sk')['title'], - upstream['templates']['system_restore_completed']['title'].format(**data)) - def test_spanish_restore_uses_maintainer_guests_terminology(self): import importlib.util spec = importlib.util.spec_from_file_location('isolated_spanish_restore', SCRIPTS / 'notification_templates.py') @@ -256,6 +209,7 @@ class OutcomeWording(unittest.TestCase): from typing import Dict, Optional ns = {'re': re, 'Dict': Dict, 'Optional': Optional, 'Any': Any, 'runtime_message': lambda key, lang, **kw: key} + extract(SCRIPTS / 'notification_templates.py', '_parse_vzdump_table', namespace=ns) parser = extract(SCRIPTS / 'notification_templates.py', '_parse_vzdump_message', namespace=ns) formatter = extract(SCRIPTS / 'notification_templates.py', '_format_vzdump_body', namespace=ns) incomplete = parser('INFO: Starting Backup of VM 104 (qemu)') @@ -392,9 +346,7 @@ class OutcomeWording(unittest.TestCase): self.catalog['runtime']['notifications']['backup'][key]).format(hostname=data['hostname']) self.assertTrue(result['title'].startswith(expected_title), result['title']) else: - source = (catalog if catalog.get('backup', {}).get('unconfirmedBody') - else self.catalog['runtime']['notifications']) - self.assertEqual(result['title'], source['templates']['backup_complete']['title'].format(hostname=data['hostname'])) + self.assertEqual(result['title'], catalog['templates']['backup_complete']['title'].format_map(module._SafeFormatDict(data))) self.assertNotIn('{hostname}', result['title']) if state == 'unconfirmed': source = (catalog if catalog.get('backup', {}).get('unconfirmedBody') @@ -407,8 +359,7 @@ class OutcomeWording(unittest.TestCase): self.assertTrue(enriched.startswith({'confirmed':'💾✅','unconfirmed':'💾❔','failed':'💾❌'}[state])) recovery = module.render_template('error_resolved', {'hostname':'node','category':'temperature', 'reason':'Old observation','duration':'3d','original_severity':'WARNING'}, lang) - recovery_source = (catalog if catalog.get('backup', {}).get('unconfirmedBody') - else self.catalog['runtime']['notifications']) + recovery_source = catalog self.assertEqual(recovery['title'], recovery_source['templates']['error_resolved']['title'].format(hostname='node',category='temperature',entity_suffix='')) self.assertNotIn('resolved', recovery['title'].lower()) if lang == 'en' else None restore = module.render_template('system_restore_completed', {'hostname':'node', 'guests':4, diff --git a/.github/scripts/tests/test_notification_pve92.py b/.github/scripts/tests/test_notification_pve92.py new file mode 100644 index 00000000..16f27487 --- /dev/null +++ b/.github/scripts/tests/test_notification_pve92.py @@ -0,0 +1,194 @@ +"""Frozen PVE 9.2 report through inert actual receiver and renderers. + +No git history, host-management import, notification send or generated fixture. +""" +import ast +import importlib.util +import re +import sys +import unittest +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[3] +SCRIPTS = ROOT / 'AppImage/scripts' +sys.path.insert(0, str(SCRIPTS)) +import notification_templates as templates + +PVE92 = '''Details +======= +VMID Name Status Time Size Filename +100 web ok 1m 1s 1 GiB vm/100/2026-09-29T17:00:00Z + +Total running time: 1m 1s +Total size: 1 GiB +''' + + +def receiver(): + tree = ast.parse((SCRIPTS / 'notification_events.py').read_text()) + owner = next(n for n in tree.body if isinstance(n, ast.ClassDef) and n.name == 'ProxmoxHookWatcher') + names = ('_classify_pve', '_map_severity', '_backup_outcome', 'process_webhook') + ns = {'re': re, 'capture_journal_context': lambda **kw: ''} + class Event: + def __init__(self, **kw): self.__dict__.update(kw); self.event_id = 'inert' + ns['NotificationEvent'] = Event + for name in names: + node = next(n for n in owner.body if isinstance(n, ast.FunctionDef) and n.name == name) + node.decorator_list = [] + exec(compile(ast.Module(body=[node], type_ignores=[]), '', 'exec'), ns) + class Queue: + def __init__(self): self.items = [] + def put(self, event): self.items.append(event) + class Receiver: + _hostname = 'node-a' + _classify_pve = ns['_classify_pve'] + _map_severity = staticmethod(ns['_map_severity']) + _backup_outcome = staticmethod(ns['_backup_outcome']) + process_webhook = ns['process_webhook'] + def __init__(self): self._queue = Queue() + return Receiver() + + +def event_for(message, severity='info'): + target = receiver() + target.process_webhook({'fields': {'type': 'vzdump'}, 'severity': severity, + 'title': 'Backup', 'message': message}) + return target._queue.items[0] + + +class PVE92Tests(unittest.TestCase): + def test_exact_maintainer_report_is_confirmed(self): + event = event_for(PVE92) + self.assertEqual(event.data['backup_outcome'], 'confirmed') + result = templates.render_template(event.event_type, event.data, 'en') + self.assertIn('Backup complete', result['title']) + self.assertIn('web (100)', result['title']) + self.assertIn('✅', result['body']) + + def test_failed_guest_identity_does_not_come_from_timestamp_or_ok_guest(self): + failed = '101 db err 1m 1s 0 B ct/101/2026-09-29T17:00:00Z' + ok = PVE92.splitlines()[3] + for rows in (ok + '\n' + failed, failed + '\n' + ok): + for severity in ('info', 'error'): + with self.subTest(rows=rows, severity=severity): + message = ('INFO: 100 01:01:06 OK\nINFO: Starting Backup of VM 100 (qemu)\n' + + PVE92.replace(ok, rows)) + event = event_for(message, severity) + self.assertEqual(event.data['backup_outcome'], 'failed') + self.assertEqual(event.data.get('vmname'), 'db') + self.assertEqual(event.data.get('vmid'), '101') + result = templates.render_template(event.event_type, event.data, 'en') + self.assertIn('db (101)', result['title']) + self.assertNotIn('web', result['title']) + self.assertNotIn('01:01:06', result['title']) + self.assertIn('❌ CT db (101)', result['body']) + + def test_pbs_prefix_is_used_in_parsed_type_and_title(self): + for prefix, kind, label in (('vm', 'qemu', 'VM'), ('ct', 'lxc', 'CT'), + ('other', '', 'VM/CT')): + with self.subTest(prefix=prefix): + message = PVE92.replace('vm/100/', prefix + '/100/') + parsed = templates._parse_vzdump_message(message) + self.assertEqual(parsed['vms'][0]['type'], kind) + event = event_for(message) + result = templates.render_template(event.event_type, event.data, 'en') + self.assertIn(label + ' web (100)', result['title']) + + def test_slovak_uses_normal_locale_resolution_without_source_sentence_overrides(self): + from unittest.mock import patch + data = {'hostname': 'node-a', 'category': 'temperature', 'entity_suffix': '', + 'reason': 'old', 'duration': '3d', 'original_severity': 'WARNING', + 'vmname': 'web', 'vmid': '100', 'storage': 'PBS', 'size': '1 GiB', + 'guests': 3, 'stubs': 0, 'stale_nodes': 0, 'components': 1, + 'warnings_block': ''} + slovak = templates._load_runtime_catalog('sk') + english = templates._load_runtime_catalog('en') + for event, field in (('error_resolved', 'title'), ('error_resolved', 'body'), + ('system_restore_completed', 'body'), + ('backup_complete', 'title'), ('backup_complete', 'body')): + with self.subTest(event=event, field=field): + value = slovak['templates'][event][field] + result = templates.render_template(event, data, 'sk') + self.assertEqual(result[field], value.format(**data)) + self.assertNotIn(value, (SCRIPTS / 'notification_templates.py').read_text()) + # Independently updated and absent leaves use the usual provider. + import copy + future = copy.deepcopy(slovak) + future['templates'][event][field] = 'REVIEWED {hostname}' + with patch.object(templates, '_load_runtime_catalog', side_effect=lambda lang: future if lang == 'sk' else english): + self.assertEqual(templates.render_template(event, data, 'sk')[field], 'REVIEWED node-a') + future['templates'][event].pop(field) + with patch.object(templates, '_load_runtime_catalog', side_effect=lambda lang: future if lang == 'sk' else english): + self.assertEqual(templates.render_template(event, data, 'sk')[field], + templates.render_template(event, data, 'en')[field]) + + def test_complete_table_beats_only_truncated_supplemental_log(self): + message = ('INFO: Starting Backup of VM 100 (qemu)\n' + PVE92 + + '\nLogs\n====\nINFO: Log output was too long to be displayed. Please see task log for details.') + self.assertEqual(event_for(message).data['backup_outcome'], 'confirmed') + for message, severity, expected in ( + (PVE92.replace('ok ', 'OK '), 'info', 'confirmed'), + (message, 'warning', 'unconfirmed'), + (message, 'error', 'failed'), + (message + '\nERROR: archive write failed', 'info', 'failed'), + (message + '\nWARNING: skipped file', 'info', 'unconfirmed'), + (PVE92.replace('ok ', 'WARNINGS '), 'info', 'unconfirmed'), + ): + with self.subTest(message=message, severity=severity): + self.assertEqual(event_for(message, severity).data['backup_outcome'], expected) + + def test_incomplete_and_unrelated_sections_cannot_certify_table(self): + for message in ( + PVE92.split('Total running time:')[0], + PVE92.replace('vm/100/2026-09-29T17:00:00Z', ''), + PVE92.replace('\n\nTotal', '\n\nLogs\n======\nTotal'), + PVE92.replace('\n\nTotal', '\n\nUnrelated section\n100 web ok\nTotal'), + PVE92.replace('1m 1s 1 GiB', 'nonsense 1 GiB'), + PVE92.replace('1 GiB vm/', 'garbage vm/'), + PVE92.replace('1m 1s 1 GiB vm/', '1m 1s'), + ): + with self.subTest(message=message): + # Complete guest logs cannot rescue a genuinely incomplete table. + message += '\nINFO: Starting Backup of VM 100 (qemu)\nINFO: Finished Backup of VM 100 (00:01:01)' + self.assertEqual(event_for(message).data['backup_outcome'], 'unconfirmed') + + def test_blank_line_between_rows_does_not_hide_a_failure(self): + message = PVE92.replace('\n\nTotal', + '\n\n101 db err 1m 1s 0 B ct/101/2026-09-29T17:00:00Z\n\nTotal') + event = event_for(message) + self.assertEqual(event.data['backup_outcome'], 'failed') + result = templates.render_template(event.event_type, event.data, 'en') + self.assertIn('db (101)', result['title']) + self.assertIn('❌ CT db (101)', result['body']) + + def test_changed_titles_and_rows_reach_actual_html_email_in_all_locales(self): + import html + from notification_channels import EmailChannel + email = object.__new__(EmailChannel) + email.subject_prefix = '[ProxMenux]' + failed = PVE92.replace('\n\nTotal', + '\n101 db err 1m 1s 0 B ct/101/2026-09-29T17:00:00Z\n\nTotal') + for lang in ('en', 'de', 'es', 'fr', 'it', 'pt', 'sk', 'sv'): + for message, severity in ((PVE92, 'info'), (failed, 'info'), (failed, 'error')): + with self.subTest(lang=lang, severity=severity, message=message): + event = event_for(message, severity) + result = templates.render_template(event.event_type, event.data, lang) + context = {**event.data, '_event_type': event.event_type, + '_notification_language': lang, '_group': result['group']} + markup = html.unescape(email._format_html(result['title'], result['body'], event.severity, context)) + self.assertIn(result['title'], markup) + if event.data['backup_outcome'] == 'failed': + self.assertIn('db (101)', result['title']) + self.assertNotIn('web', result['title']) + self.assertIn('❌ CT db (101)', markup) + status = templates.runtime_message('channels.email.status.failed', lang) + else: + self.assertIn('VM web (100)', result['title']) + self.assertIn('✅ VM web (100)', markup) + status = templates.runtime_message('channels.email.status.completed', lang) + badge = (templates.runtime_message('channels.email.severity.critical', lang) + if event.event_type == 'backup_fail' else status) + self.assertIn('>' + badge.upper() + '', markup) + + +if __name__ == '__main__': unittest.main() diff --git a/AppImage/scripts/notification_events.py b/AppImage/scripts/notification_events.py index 9e949f60..6d9ac520 100644 --- a/AppImage/scripts/notification_events.py +++ b/AppImage/scripts/notification_events.py @@ -4298,41 +4298,20 @@ class ProxmoxHookWatcher: return 'failed' starts = re.findall(r'(?im)\bStarting Backup of VM (\d+)\s*\(', text) finished = re.findall(r'(?im)\bFinished Backup of VM (\d+)\s*\(', text) - lines = text.splitlines() - table_outcome = None - for index, header in enumerate(lines): - if not re.match(r'\s*VMID\s+Name\s+Status\b', header, re.IGNORECASE): - continue - status_start = header.find('Status') - status_end = header.find('Time', status_start) - if status_start < 0 or status_end < 0: - break - rows = [] - for line in lines[index + 1:]: - if re.match(r'\s*Total\b', line, re.IGNORECASE): - table_outcome = ('confirmed' if rows and all(status == 'OK' for status in rows) - else 'unconfirmed') - break - if not line.strip(): - break - if not re.match(r'\s*\d+\s+', line): - break - status = line[status_start:status_end].strip().upper() - if status in ('ERROR', 'ERR'): - return 'failed' - rows.append(status) - break + from notification_templates import _parse_vzdump_table + table = _parse_vzdump_table(text) + if table is not None and any(guest['status'].lower() == 'error' for guest in table['vms']): + return 'failed' if severity not in ('info', 'ok', 'success') or re.search( r'(?im)(?:^\s*WARNING:|\bWARNINGS\s*:\s*\d+)', text): return 'unconfirmed' # A present table is authoritative: do not certify an incomplete table # from a finished guest log, or reject a complete OK table merely # because the extra diagnostic log was truncated before its finishes. - if table_outcome is not None: - return table_outcome - if any(re.match(r'\s*VMID\s+Name\s+Status\b', line, re.IGNORECASE) - for line in lines): - return 'unconfirmed' + if table is not None: + return ('confirmed' if table['complete'] and + all(guest['status'].lower() == 'ok' for guest in table['vms']) + else 'unconfirmed') if starts: return 'confirmed' if sorted(starts) == sorted(finished) else 'unconfirmed' if re.search( @@ -4498,10 +4477,17 @@ class ProxmoxHookWatcher: if vmids: data['vmid'] = vmids[0] entity_id = vmids[0] - # Try to extract VM name from the table line - name_m = re.search(r'(\d+)\s+(\S+)\s+(?:OK|ERROR|WARNINGS)', message) - if name_m: - data['vmname'] = name_m.group(2) + from notification_templates import _parse_vzdump_message + parsed = _parse_vzdump_message(message) or {} + guests = parsed.get('vms', []) + if data.get('backup_outcome') == 'failed': + guests = [guest for guest in guests if guest.get('status', '').lower() == 'error'] + if len(guests) == 1: + data['vmid'] = guests[0]['vmid'] + data['vmname'] = guests[0]['name'] + else: + # Do not make one successful guest the subject of a batch failure. + data.pop('vmid', None) # Extract size from "Total size: X" size_m = re.search(r'Total size:\s*(.+?)(?:\n|$)', message) if size_m: diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index eba823a9..d0a20dd5 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -206,6 +206,51 @@ def _format_lxc_update_details(data: Dict[str, Any], language: str) -> str: # ─── vzdump message parser ─────────────────────────────────────── +def _parse_vzdump_table(message: str) -> Optional[Dict[str, Any]]: + """Read the bounded fixed-column summary for both outcomes and guest details.""" + lines = message.splitlines() + for index, header in enumerate(lines): + if not re.match(r'\s*VMID\s+Name\s+Status\b', header, re.IGNORECASE): + continue + columns = [re.search(r'\b' + name + r'\b', header, re.IGNORECASE) + for name in ('VMID', 'Name', 'Status', 'Time', 'Size', 'Filename')] + if not all(columns): + return {'vms': [], 'complete': False} + starts = [column.start() for column in columns if column is not None] + if starts != sorted(starts): + return {'vms': [], 'complete': False} + rows = [] + valid = True + complete = False + for line in lines[index + 1:]: + if not line.strip(): + continue + if re.match(r'\s*Total running time:\s*\S', line, re.IGNORECASE): + complete = valid and bool(rows) + break + # Blanks are allowed, but no unrelated section can extend the table. + if not re.match(r'\s*\d+\s+', line): + break + values = [line[a:b].strip() for a, b in + zip(starts, starts[1:] + [len(line)])] + vmid, name, status, duration, size, filename = values + if not vmid.isdigit(): + valid = False + break + valid = bool(valid and all(values) + and re.fullmatch(r'(?:\d+:\d{2}:\d{2}|(?:\d+[dhms]\s*)+)', duration) + and re.fullmatch(r'\d+(?:\.\d+)?\s*(?:[KMGTPE]i?B?|B)', size, re.IGNORECASE)) + if status.lower() in ('err', 'error'): + status = 'error' + kind = ('lxc' if 'lxc' in filename or filename.startswith('ct/') else + 'qemu' if 'qemu' in filename or filename.startswith('vm/') else '') + rows.append({'vmid': vmid, 'name': name, 'status': status, + 'time': duration, 'size': size, 'filename': filename, + 'type': kind}) + return {'vms': rows, 'complete': bool(complete)} + return None + + def _parse_vzdump_message(message: str) -> Optional[Dict[str, Any]]: """Parse a PVE vzdump notification message into structured data. @@ -225,55 +270,10 @@ def _parse_vzdump_message(message: str) -> Optional[Dict[str, Any]]: lines = message.split('\n') - # ── Strategy 1: classic table (local/NFS/CIFS storage) ── - header_idx = -1 - for i, line in enumerate(lines): - if re.match(r'\s*VMID\s+Name\s+Status', line, re.IGNORECASE): - header_idx = i - break - - if header_idx >= 0: - # Use column positions from the header to slice each row. - # Header: "VMID Name Status Time Size Filename" - header = lines[header_idx] - col_starts = [] - for col_name in ['VMID', 'Name', 'Status', 'Time', 'Size', 'Filename']: - idx = header.find(col_name) - if idx >= 0: - col_starts.append(idx) - - if len(col_starts) == 6: - for line in lines[header_idx + 1:]: - stripped = line.strip() - if not stripped or stripped.startswith('Total') or stripped.startswith('Logs') or stripped.startswith('='): - break - # Pad line to avoid index errors - padded = line.ljust(col_starts[-1] + 50) - vmid = padded[col_starts[0]:col_starts[1]].strip() - name = padded[col_starts[1]:col_starts[2]].strip() - status = padded[col_starts[2]:col_starts[3]].strip() - if status.lower() in ('err', 'error'): - status = 'error' - time_val = padded[col_starts[3]:col_starts[4]].strip() - size = padded[col_starts[4]:col_starts[5]].strip() - filename = padded[col_starts[5]:].strip() - - if vmid and vmid.isdigit(): - # Infer type from filename (vzdump-lxc-NNN or vzdump-qemu-NNN) - vm_type = '' - if 'lxc' in filename: - vm_type = 'lxc' - elif 'qemu' in filename: - vm_type = 'qemu' - vms.append({ - 'vmid': vmid, - 'name': name, - 'status': status, - 'time': time_val, - 'size': size, - 'filename': filename, - 'type': vm_type, - }) + # The same summary rows drive classification, rich bodies and identities. + table = _parse_vzdump_table(message) + if table is not None: + vms = table['vms'] # ── Strategy 2: log-style (PBS / Proxmox Backup Server) ── # Parse from the full vzdump log lines. @@ -1856,19 +1856,6 @@ def render_template(event_type: str, data: Dict[str, Any], _catalog_value(requested_catalog, key) or _catalog_value(english_catalog, key) ) - # Keep the Slovak catalog with its maintainer. Suppress only the - # exact stale report leaves, not future translations or safe titles. - # An unrelated outcome key cannot version recovery/restore wording. - stale_slovak_reports = { - 'templates.backup_complete.title': '{hostname} → {storage}: Záloha dokončená — {vmname} ({vmid})', - 'templates.backup_complete.body': 'Záloha {vmname} (ID: {vmid}) na úložisku {storage} bola úspešne dokončená.\nVeľkosť: {size}', - 'templates.error_resolved.title': '{hostname}: Vyriešené - {category}{entity_suffix}', - 'templates.error_resolved.body': 'Problém v kategórii {category} bol vyriešený.\n{reason}\n🚦 Predchádzajúca závažnosť: {original_severity}\n⏱️ Trvanie: {duration}', - 'templates.system_restore_completed.body': 'Úlohy po obnove boli dokončené na pozadí.\n\nPoužité VM a LXC: {guests}\nZástupné priečinky bind mountov: {stubs}\nOdstránené zastarané priečinky uzlov: {stale_nodes}\nPreinštalované súčasti: {components}\nTrvanie: {duration}\n{warnings_block}\nUzol je teraz úplne pripravený na použitie.', - } - if (requested_language == 'sk' and key in stale_slovak_reports - and localized == stale_slovak_reports[key]): - localized = _catalog_value(english_catalog, key) if localized: template[field] = localized backup_title_target = '' @@ -1877,25 +1864,28 @@ def render_template(event_type: str, data: Dict[str, Any], if outcome == 'confirmed': template['title'] = runtime_message('backup.confirmedTitle', language, hostname=data.get('hostname') or _get_hostname()) - parsed_backup = _parse_vzdump_message(str(data.get('pve_message') or '')) - storage = str((parsed_backup or {}).get('storage_name') or data.get('storage') or '').strip() - guests = (parsed_backup or {}).get('vms') or [] - target = [] - if storage: - target.append(storage) - if len(guests) == 1: - guest = guests[0] - kind = 'VM' if guest.get('type') == 'qemu' else 'CT' if guest.get('type') == 'lxc' else 'VM/CT' - name = guest.get('name') or kind - target.append(f"{kind} {name} ({guest['vmid']})" if name != kind - else f"{kind} {guest['vmid']}") - if target: - backup_title_target = ' — ' + ' · '.join(target) template['body'] = runtime_message('backup.confirmedBody', language) elif outcome == 'failed': template['title'] = runtime_message('backup.errorTitle', language, hostname=data.get('hostname') or _get_hostname()) template['body'] = runtime_message('backup.errorBody', language) + if event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'failed'): + parsed_backup = _parse_vzdump_message(str(data.get('pve_message') or '')) + storage = str((parsed_backup or {}).get('storage_name') or data.get('storage') or '').strip() + guests = (parsed_backup or {}).get('vms') or [] + if data.get('backup_outcome') == 'failed': + guests = [guest for guest in guests if guest.get('status', '').lower() == 'error'] + target = [] + if storage: + target.append(storage) + if len(guests) == 1: + guest = guests[0] + kind = 'VM' if guest.get('type') == 'qemu' else 'CT' if guest.get('type') == 'lxc' else 'VM/CT' + name = guest.get('name') or kind + target.append(f"{kind} {name} ({guest['vmid']})" if name != kind + else f"{kind} {guest['vmid']}") + if target: + backup_title_target = ' — ' + ' · '.join(target) # Ensure hostname is always available variables = { From e24308c0172304a5f56ecf8e35282c76009d2f8d Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Tue, 29 Sep 2026 23:52:07 +0200 Subject: [PATCH 06/14] Correct backup outcome diagnostics and email presentation chains --- .github/scripts/tests/notification_fixture.py | 51 ++++ .../tests/test_notification_corrections.py | 251 ++++++++++++++++++ .../test_notification_outcome_wording.py | 8 +- AppImage/scripts/notification_channels.py | 25 +- AppImage/scripts/notification_events.py | 21 +- AppImage/scripts/notification_manager.py | 5 + AppImage/scripts/notification_templates.py | 80 +++++- .../tests/test_notification_runtime_i18n.py | 13 +- .../scripts/tests/test_vzdump_ai_integrity.py | 8 +- 9 files changed, 431 insertions(+), 31 deletions(-) create mode 100644 .github/scripts/tests/notification_fixture.py create mode 100644 .github/scripts/tests/test_notification_corrections.py diff --git a/.github/scripts/tests/notification_fixture.py b/.github/scripts/tests/notification_fixture.py new file mode 100644 index 00000000..90ba64e4 --- /dev/null +++ b/.github/scripts/tests/notification_fixture.py @@ -0,0 +1,51 @@ +"""Assertion-free inert actual consumers; no operational host imports.""" +import ast +import re +import sys +from pathlib import Path +ROOT = Path(__file__).resolve().parents[3] +SCRIPTS = ROOT / 'AppImage/scripts' +if str(SCRIPTS) not in sys.path: + sys.path.insert(0, str(SCRIPTS)) +import notification_templates as templates +# Display-name resolution is an infrastructure boundary, never load manager. +templates._get_hostname = lambda: 'node-a' +from notification_channels import EmailChannel +LANGUAGES = ('en', 'de', 'es', 'fr', 'it', 'pt', 'sk', 'sv') + +def extract(path, name, owner, ns): + tree = ast.parse(path.read_text()) + nodes = tree.body if owner is None else next(n.body for n in tree.body if isinstance(n, ast.ClassDef) and n.name == owner) + node = next(n for n in nodes if isinstance(n, ast.FunctionDef) and n.name == name) + node.decorator_list = [] + exec(compile(ast.Module(body=[node], type_ignores=[]), str(path), 'exec'), ns) + return ns[name] + +def receive(message, severity='info', title='Backup', kind='vzdump'): + ns = {'re': re, 'capture_journal_context': lambda **kw: ''} + class Event: + def __init__(self, **kw): self.__dict__.update(kw); self.event_id = 'inert' + ns['NotificationEvent'] = Event + methods = {name: extract(SCRIPTS / 'notification_events.py', name, 'ProxmoxHookWatcher', ns) + for name in ('_classify_pve', '_map_severity', '_backup_outcome', 'process_webhook')} + class Queue: + def __init__(self): self.items = [] + def put(self, event): self.items.append(event) + class Receiver: + _hostname = 'node-a' + _classify_pve = methods['_classify_pve'] + _map_severity = staticmethod(methods['_map_severity']) + _backup_outcome = staticmethod(methods['_backup_outcome']) + process_webhook = methods['process_webhook'] + def __init__(self): self._queue = Queue() + target = Receiver() + result = target.process_webhook({'fields': {'type': kind}, 'severity': severity, 'title': title, 'message': message}) + assert result['accepted'] and len(target._queue.items) == 1 + return target._queue.items[0] + +def email(event_type, data, severity='INFO', language='en'): + result = templates.render_template(event_type, data, language) + channel = object.__new__(EmailChannel) + channel.subject_prefix = '[ProxMenux]' + context = {**data, 'severity': severity, '_event_type': event_type, '_group': result['group'], '_notification_language': language} + return result, channel._format_html(result['title'], result['body'], severity, context) diff --git a/.github/scripts/tests/test_notification_corrections.py b/.github/scripts/tests/test_notification_corrections.py new file mode 100644 index 00000000..94ae1469 --- /dev/null +++ b/.github/scripts/tests/test_notification_corrections.py @@ -0,0 +1,251 @@ +"""Whole-PR outcome corrections, inert producer/actual email consumers.""" +import html +import unittest +from notification_fixture import templates, receive, email, LANGUAGES + +REPORT = """Details +======= +VMID Name Status Time Size Filename +100 web ok 1m 1s 1 GiB vm/100/2026-09-29T17:00:00Z + +Total running time: 1m 1s +Total size: 1 GiB +""" + +class CorrectionTests(unittest.TestCase): + def test_reversed_finish_is_not_completion_evidence(self): + event = receive('INFO: Finished Backup of VM 100 (00:01:01)\nINFO: Starting Backup of VM 100 (qemu)') + self.assertEqual(event.data['backup_outcome'], 'unconfirmed') + + + def test_interleaved_complete_logs_keep_both_finished_guests(self): + message = ('INFO: Starting Backup of VM 100 (qemu)\nINFO: VM Name: web\n' + 'INFO: Starting Backup of VM 101 (lxc)\nINFO: CT Name: db\n' + 'INFO: Finished Backup of VM 100 (00:01:01)\nINFO: Finished Backup of VM 101 (00:01:02)') + event = receive(message) + self.assertEqual(event.data['backup_outcome'], 'confirmed') + result, markup = email(event.event_type, event.data, event.severity) + self.assertIn('✅ VM web (100)', result['body']) + self.assertIn('✅ CT db (101)', result['body']) + self.assertNotIn('❔', result['body']) + self.assertIn('00:01:01', result['body']) + + + def test_backup_identity_has_event_scoped_mail_compatible_wrapping(self): + result, markup = email('backup_complete', {'hostname': 'n' * 64, + 'backup_outcome': 'confirmed', 'pve_message': 'INFO: Starting Backup of VM 100 (qemu)\nINFO: VM Name: customerproductionpostgresqlreplicaeuropewestdatacenter01\nINFO: Finished Backup of VM 100 (00:01:01)'}) + self.assertIn('table-layout:fixed;', markup) + title_tag = markup.split('

RESOLVED', markup) + self.assertNotIn('color:#16a34a', markup) + self.assertEqual(data['_event_type'], 'node_reconnect') # caller not mutated + + + def test_disappearance_body_keeps_observation_age_not_green_severity(self): + data = {'hostname': 'node-a', 'category': 'temperature', 'reason': 'old observation', + 'duration': '3d 2h', 'original_severity': 'WARNING', 'severity': 'OK'} + for language in LANGUAGES: + result, markup = email('error_resolved', data, 'OK', language) + for line in result['body'].splitlines(): + if line.strip(): self.assertIn(line.strip(), html.unescape(markup)) + self.assertNotIn('>OK', markup) + self.assertNotIn('color:#16a34a', markup) + self.assertNotIn('>RESOLVED', markup) + + + def test_actual_restore_endpoint_warnings_and_counts_reach_email(self): + from notification_fixture import extract, SCRIPTS + from types import SimpleNamespace + for warnings in ('', 'missing module zfs'): + events = [] + ns = {'request': SimpleNamespace(remote_addr='127.0.0.1', get_json=lambda **kw: { + 'hostname': 'node-a', 'guests': '3', 'stubs': '1', 'stale_nodes': '0', + 'components': '2', 'duration': '2m', 'warnings': warnings}), + 'notification_manager': SimpleNamespace(emit_event=lambda **kw: events.append(kw)), + 'jsonify': lambda value: value} + handler = extract(SCRIPTS / 'flask_notification_routes.py', 'internal_restore_event', None, ns) + response, status = handler() + self.assertEqual(status, 200) + event = events[0] + for language in LANGUAGES: + result, markup = email(event['event_type'], event['data'], event['severity'], language) + self.assertIn('2m', markup) + for line in result['body'].splitlines(): + if line.strip(): self.assertIn(line.strip(), html.unescape(markup)) + if warnings: self.assertIn(warnings, markup) + + + def test_raw_display_hostname_is_substituted_exactly_once(self): + for event_type, outcome in (('backup_complete', 'confirmed'), ('backup_complete', 'failed'), ('backup_fail', 'failed')): + for hostname in ('Sala {rack} – Zürich', 'node-{vmid}', 'Sala {rack.location}'): + for language in LANGUAGES: + result, markup = email(event_type, {'hostname': hostname, + 'backup_outcome': outcome, 'pve_message': REPORT}, language=language) + self.assertIn(hostname, result['title']) + self.assertIn(hostname, html.unescape(markup)) + + + def test_malformed_numeric_report_is_queued_uncertain_and_renderable(self): + for size in ('1..5 GiB', '..5 GiB'): + message = REPORT.replace('1 GiB ', size.ljust(9)).split('Total size:')[0] + event = receive(message) + self.assertEqual(event.data['backup_outcome'], 'unconfirmed') + result, markup = email(event.event_type, event.data, event.severity) + self.assertIn(size, result['body']) + + + def test_unknown_backup_type_keeps_explicit_err_but_cannot_certify_ok(self): + event = receive(REPORT.replace('ok ', 'err '), kind='') + self.assertEqual(event.event_type, 'backup_complete') + self.assertEqual(event.severity, 'INFO') + self.assertEqual(event.data['backup_outcome'], 'failed') + result, markup = email(event.event_type, event.data, event.severity) + self.assertIn('FAILED', markup) + self.assertEqual(receive(REPORT, kind='').data['backup_outcome'], 'unconfirmed') + + + def test_confirmed_metadata_only_context_is_retained(self): + data = {'hostname': 'node-a', 'backup_outcome': 'confirmed', + 'vmid': '100', 'vmname': 'web {literal}', 'storage': 'PBS', 'size': '1 GiB'} + for language in LANGUAGES: + result, markup = email('backup_complete', data, language=language) + self.assertIn('web {literal} (100)', result['title']) + self.assertIn('1 GiB', result['body']) + self.assertIn('1 GiB', markup) + + + def test_explicit_guest_failure_overrides_only_the_linked_ok_row(self): + other = '101 db ok 1m 1s 1 GiB ct/101/2026-09-29T17:00:00Z' + message = REPORT.replace('\n\nTotal', '\n' + other + '\n\nTotal') + diagnostic = '100: 2026-09-29 17:00:00 ERROR: Backup of VM 100 failed - archive write failed' + for severity in ('info', 'error'): + event = receive(message + '\n' + diagnostic, severity) + self.assertEqual(event.data.get('vmid'), '100') + for language in LANGUAGES: + result, markup = email(event.event_type, event.data, event.severity, language) + self.assertIn('❌ VM web (100)', result['body']) + self.assertNotIn('✅ VM web (100)', markup) + self.assertIn('✅ CT db (101)', result['body']) + self.assertIn('web (100)', result['title']) + + + def test_official_week_month_year_durations_stay_confirmed(self): + for duration in ('1w', '1w 1m 1s', '1M', '1y'): + # Fixed-column widths are unchanged for these bounded values. + message = REPORT.replace('1m 1s ', duration.ljust(9)) + event = receive(message) + self.assertEqual(event.data['backup_outcome'], 'confirmed', duration) + result, markup = email(event.event_type, event.data, event.severity) + self.assertIn(duration, result['body']) + + + def test_prefixed_error_diagnostics_survive_both_source_severities(self): + diagnostic = '100: 2026-09-29 17:00:00 ERROR: archive write failed: permission denied' + for severity in ('info', 'error'): + event = receive(REPORT + '\n' + diagnostic, severity) + self.assertEqual(event.data['backup_outcome'], 'failed') + for language in LANGUAGES: + result, markup = email(event.event_type, event.data, event.severity, language) + self.assertIn(diagnostic, result['body']) + self.assertIn(diagnostic, html.unescape(markup)) + # A job-level error must not invent a failed guest. + self.assertIn('✅ VM web (100)', result['body']) + + + def test_official_prefixed_warning_is_uncertain_and_retained(self): + warning = '100: 2026-09-29 17:00:00 WARN: unable to add notes - permission denied' + event = receive(REPORT + '\n' + warning) + self.assertEqual(event.data['backup_outcome'], 'unconfirmed') + for language in LANGUAGES: + result, markup = email(event.event_type, event.data, event.severity, language) + self.assertIn(warning, result['body']) + self.assertIn(warning, html.unescape(markup)) + + def test_batch_and_abort_failure_titles_have_no_empty_guest_slot(self): + failed = REPORT.replace('ok ', 'err ') + failed = failed.replace('\n\nTotal', '\n101 db err 1m 1s 0 B null\n\nTotal') + for message in (failed, REPORT.replace('ok ', 'todo '), + REPORT.split('100 web')[0] + '\nTotal running time: 0s'): + event = receive(message + '\nINFO: vzdump --storage PBS', 'error') + for language in LANGUAGES: + result, markup = email(event.event_type, event.data, event.severity, language) + self.assertNotIn('()', result['title']) + self.assertIn('PBS', result['title']) + self.assertNotIn('web (100)', result['title']) + + def test_subject_only_setup_failure_survives_actual_email(self): + reason = 'unable to activate storage PBS' + title = 'vzdump backup status (node-a): backup failed: ' + reason + message = REPORT.split('100 web')[0] + '\nTotal running time: 0s\nTotal size: 0 B' + event = receive(message, 'error', title) + for language in LANGUAGES: + result, markup = email(event.event_type, event.data, event.severity, language) + self.assertIn(reason, result['body']) + self.assertIn(reason, html.unescape(markup)) + +if __name__ == '__main__': unittest.main() diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py index 5b2f8820..5be667c3 100644 --- a/.github/scripts/tests/test_notification_outcome_wording.py +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -10,6 +10,7 @@ import unittest from pathlib import Path from typing import Any from unittest.mock import patch +from notification_fixture import templates as actual_templates ROOT = Path(__file__).resolve().parents[3] SCRIPTS = ROOT / 'AppImage/scripts' @@ -292,10 +293,9 @@ class OutcomeWording(unittest.TestCase): build = extract(path, '_build_detail_rows', 'EmailChannel', ns) fmt = extract(path, '_format_html', 'EmailChannel', ns) class Email: - _SEV_STYLE = {'OK': {'color':'#16a34a','bg':'#f0fdf4','border':'#bbf7d0'}, - 'CRITICAL': {'color':'#dc2626','bg':'#fef2f2','border':'#fecaca'}, - 'INFO': {'color':'blue','bg':'white','border':'gray'}} - _SEV_DEFAULT = {'color':'#6b7280','bg':'#f9fafb','border':'#e5e7eb'} + from notification_channels import EmailChannel + _SEV_STYLE = EmailChannel._SEV_STYLE + _SEV_DEFAULT = EmailChannel._SEV_DEFAULT subject_prefix = 'ProxMenux' _build_detail_rows = staticmethod(build) badge = catalog['channels']['email']['severity'].get('observation') or english['channels']['email']['severity']['observation'] diff --git a/AppImage/scripts/notification_channels.py b/AppImage/scripts/notification_channels.py index 6580826f..eddc2bde 100644 --- a/AppImage/scripts/notification_channels.py +++ b/AppImage/scripts/notification_channels.py @@ -1053,11 +1053,13 @@ class EmailChannel(NotificationChannel): status = 'unconfirmed' sev['label'] = _runtime_text(f'email.status.{status}', data) group = data.get('_group', 'other') - # Keep unbroken recorded text inside the temperature email's table. - # Both properties are inline for mail clients; other events retain - # their original markup and layout. - temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if event_type == 'temp_high' else '' - temp_table_layout = 'table-layout:fixed;' if event_type == 'temp_high' else '' + # Scoped inline mail-compatible wrapping: temperature measurements + # and backup identities/raw diagnostics. Other events retain layout. + backup_email = event_type in {'backup_complete', 'backup_fail'} + temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if event_type == 'temp_high' or backup_email else '' + temp_table_layout = 'table-layout:fixed;' if event_type == 'temp_high' or backup_email else '' + backup_title_wrap = temp_cell_wrap if backup_email else '' + backup_metadata_layout = 'table-layout:fixed;' if backup_email else '' section_label = _runtime_text(f'email.groups.{group}', data) report_label = _runtime_text('email.report', data, group=section_label) host_label = _runtime_text('email.host', data) @@ -1089,6 +1091,13 @@ class EmailChannel(NotificationChannel): for line in body.split('\n') if line.strip() ) + if event_type in {'system_restore_completed', 'error_resolved'}: + # Observation age/disappearance must not become a green OK row. + # The endpoint's warnings_block and task counts live in the + # localized body, not the generic services Event row. + detail_rows = [('', html_mod.escape(line.strip())) + for line in body.split('\n') if line.strip()] + # ── Fallback: if no structured rows, render body text lines ── if not detail_rows: for line in body.split('\n'): @@ -1155,15 +1164,15 @@ class EmailChannel(NotificationChannel):
-

{html_mod.escape(display_title)}

+

{html_mod.escape(display_title)}

- +
-
+ {html_mod.escape(host_label)}: {html_mod.escape(data.get('hostname', ''))} diff --git a/AppImage/scripts/notification_events.py b/AppImage/scripts/notification_events.py index 6d9ac520..4aa91067 100644 --- a/AppImage/scripts/notification_events.py +++ b/AppImage/scripts/notification_events.py @@ -4294,7 +4294,7 @@ class ProxmoxHookWatcher: """Distinguish explicit failure, complete guest logs and unknown results.""" text = str(message or '') if severity in ('error', 'err', 'critical') or re.search( - r'(?im)^\s*(?:ERROR:|TASK ERROR:|.*\bStatus\s+ERROR\b)', text): + r'(?im)^\s*(?:\d+:\s*)?(?:\d{4}-\d{2}-\d{2}\s+\S+\s+)?(?:ERROR:|TASK ERROR:|.*\bStatus\s+ERROR\b)', text): return 'failed' starts = re.findall(r'(?im)\bStarting Backup of VM (\d+)\s*\(', text) finished = re.findall(r'(?im)\bFinished Backup of VM (\d+)\s*\(', text) @@ -4303,7 +4303,7 @@ class ProxmoxHookWatcher: if table is not None and any(guest['status'].lower() == 'error' for guest in table['vms']): return 'failed' if severity not in ('info', 'ok', 'success') or re.search( - r'(?im)(?:^\s*WARNING:|\bWARNINGS\s*:\s*\d+)', text): + r'(?im)(?:^\s*(?:\d+:\s*)?(?:\d{4}-\d{2}-\d{2}\s+\S+\s+)?WARN(?:ING)?:|\bWARNINGS\s*:\s*\d+)', text): return 'unconfirmed' # A present table is authoritative: do not certify an incomplete table # from a finished guest log, or reject a complete OK table merely @@ -4313,7 +4313,16 @@ class ProxmoxHookWatcher: all(guest['status'].lower() == 'ok' for guest in table['vms']) else 'unconfirmed') if starts: - return 'confirmed' if sorted(starts) == sorted(finished) else 'unconfirmed' + pending = {} + for match in re.finditer(r'(?im)\b(Starting|Finished) Backup of VM (\d+)\s*\(', text): + action, vmid = match.groups() + if action.lower() == 'starting': + pending[vmid] = pending.get(vmid, 0) + 1 + elif not pending.get(vmid): + return 'unconfirmed' # A finish before its start is not evidence. + else: + pending[vmid] -= 1 + return 'confirmed' if not any(pending.values()) else 'unconfirmed' if re.search( r'(?im)^\s*(?:INFO:\s*)?TASK OK\s*$', text): return 'confirmed' @@ -4379,10 +4388,10 @@ class ProxmoxHookWatcher: } if event_type in ('backup_complete', 'backup_fail'): # This is presentation metadata, not a new event/toggle/delivery path. + outcome = self._backup_outcome(severity_raw, message) data['backup_outcome'] = ( - 'failed' if event_type == 'backup_fail' else - self._backup_outcome(severity_raw, message) if pve_type == 'vzdump' - else 'unconfirmed' + 'failed' if event_type == 'backup_fail' or outcome == 'failed' else + outcome if pve_type == 'vzdump' else 'unconfirmed' ) if pve_type == 'replication': diff --git a/AppImage/scripts/notification_manager.py b/AppImage/scripts/notification_manager.py index 2b68d298..332072a9 100644 --- a/AppImage/scripts/notification_manager.py +++ b/AppImage/scripts/notification_manager.py @@ -2467,6 +2467,11 @@ class NotificationManager: runtime_data.get('hostname'), self._config, ) runtime_data.setdefault('_notification_language', self._notification_language()) + # Match queued dispatch's presentation context for these outcome + # notices; this does not alter event/severity or direct-send policy. + if event_type in ('backup_complete', 'backup_fail', 'error_resolved', 'system_restore_completed'): + runtime_data['_event_type'] = event_type + runtime_data['_group'] = TEMPLATES[event_type].get('group', 'other') # Render template if available if event_type in TEMPLATES and not message: diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index d0a20dd5..69d6bdbd 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -238,7 +238,7 @@ def _parse_vzdump_table(message: str) -> Optional[Dict[str, Any]]: valid = False break valid = bool(valid and all(values) - and re.fullmatch(r'(?:\d+:\d{2}:\d{2}|(?:\d+[dhms]\s*)+)', duration) + and re.fullmatch(r'(?:\d+:\d{2}:\d{2}|(?:\d+[yMwdhms]\s*)+)', duration) and re.fullmatch(r'\d+(?:\.\d+)?\s*(?:[KMGTPE]i?B?|B)', size, re.IGNORECASE)) if status.lower() in ('err', 'error'): status = 'error' @@ -309,6 +309,18 @@ def _parse_vzdump_message(message: str) -> Optional[Dict[str, Any]]: } continue + # A finish can belong to a guest already stored when another + # start arrived. Preserve that guest's actual completion too. + prior_finish = re.match(r'Finished Backup of VM (\d+)\s+\(([^)]+)\)', clean) + if prior_finish: + prior = next((vm for vm in reversed(vms) + if vm['vmid'] == prior_finish.group(1)), None) + if prior is not None: + prior['time'] = prior_finish.group(2) + if prior['status'] != 'error': + prior['status'] = 'ok' + continue + if current_vm: # Guest name m_name = re.match(r'(?:CT|VM) Name:\s*(.+)', clean) @@ -357,6 +369,21 @@ def _parse_vzdump_message(message: str) -> Optional[Dict[str, Any]]: if current_vm: vms.append(current_vm) + # Explicit guest-linked failures outrank a contradictory summary OK row. + # Job-level/prune errors do not invalidate unrelated successfully saved guests. + for line in lines: + error = re.match(r'^\s*(?:\d+:\s*)?(?:\d{4}-\d{2}-\d{2}\s+\S+\s+)?(?:ERROR:|TASK ERROR:)\s*(.*)', + line, re.IGNORECASE) + if not error: + continue + failed_guest = re.search(r'\bBackup of (?:VM|CT) (\d+) failed\b|\bbackup failed for (?:VM|CT) (\d+)\b', + error.group(1), re.IGNORECASE) + if failed_guest: + vmid = failed_guest.group(1) or failed_guest.group(2) + for vm in vms: + if vm['vmid'] == vmid: + vm['status'] = 'error' + # ── Extract totals ── for line in lines: m_time = re.search(r'Total running time:\s*(.+)', line) @@ -372,7 +399,7 @@ def _parse_vzdump_message(message: str) -> Optional[Dict[str, Any]]: sizes_gib = 0.0 for vm in vms: s = vm.get('size', '') - m = re.match(r'([\d.]+)\s+(.*)', s) + m = re.fullmatch(r'(\d+(?:\.\d+)?)\s+([KMGTPE]i?B|B)', s, re.IGNORECASE) if m: val = float(m.group(1)) unit = m.group(2).strip().upper() @@ -1869,11 +1896,19 @@ def render_template(event_type: str, data: Dict[str, Any], template['title'] = runtime_message('backup.errorTitle', language, hostname=data.get('hostname') or _get_hostname()) template['body'] = runtime_message('backup.errorBody', language) - if event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'failed'): + if event_type == 'backup_fail': + template['title'] = runtime_message('backup.errorTitle', language, + hostname=data.get('hostname') or _get_hostname()) + if event_type == 'backup_fail' or (event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'failed')): parsed_backup = _parse_vzdump_message(str(data.get('pve_message') or '')) storage = str((parsed_backup or {}).get('storage_name') or data.get('storage') or '').strip() guests = (parsed_backup or {}).get('vms') or [] - if data.get('backup_outcome') == 'failed': + # Explicit confirmed manual metadata is useful context, not evidence + # about an unparsed batch. Only use it when there is no raw report. + if not data.get('pve_message') and data.get('backup_outcome') == 'confirmed' and data.get('vmid'): + guests = [{'vmid': str(data['vmid']), 'name': str(data.get('vmname') or ''), + 'type': str(data.get('vm_type') or ''), 'status': 'ok'}] + if event_type == 'backup_fail' or data.get('backup_outcome') == 'failed': guests = [guest for guest in guests if guest.get('status', '').lower() == 'error'] target = [] if storage: @@ -1915,6 +1950,11 @@ def render_template(event_type: str, data: Dict[str, Any], 'log_file': '', } variables.update(data) + if event_type == 'backup_fail' or (event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'failed')): + # The provider has already substituted raw Display Names. Insert the + # resolved title as a value, never reinterpret its literal braces. + variables['_backup_title'] = template['title'] + template['title'] = '{_backup_title}' # Old persisted errors and manual events may lack a complete reading. # Accept plain numeric strings, but never interpret booleans or objects as @@ -2047,11 +2087,11 @@ def render_template(event_type: str, data: Dict[str, Any], if parsed: is_success = (event_type == 'backup_complete') body_text = _format_vzdump_body(parsed, is_success, language=language) - if event_type == 'backup_complete' and data.get('backup_outcome') == 'failed': - error_lines = [line.strip() for line in pve_message.splitlines() - if re.match(r'^\s*(?:ERROR:|TASK ERROR)', line, re.IGNORECASE)] - if error_lines: - body_text += '\n' + '\n'.join(error_lines) + diagnostic_lines = [line.strip() for line in pve_message.splitlines() + if re.match(r'^\s*(?:\d+:\s*)?(?:\d{4}-\d{2}-\d{2}\s+\S+\s+)?(?:WARN(?:ING)?:|ERROR:|TASK ERROR)', + line, re.IGNORECASE)] + if diagnostic_lines: + body_text += '\n' + '\n'.join(dict.fromkeys(diagnostic_lines)) else: # Couldn't parse -- use PVE raw message as body body_text = pve_message.strip() @@ -2068,6 +2108,28 @@ def render_template(event_type: str, data: Dict[str, Any], except (ValueError, IndexError): body_text = template['body'] + if event_type == 'backup_complete' and data.get('backup_outcome') == 'confirmed' and not pve_message: + context = [] + if data.get('vmid'): + name = str(data.get('vmname') or '') + context.append(f"{name} ({data['vmid']})" if name else str(data['vmid'])) + if data.get('storage'): + context.append(str(data['storage'])) + if data.get('size'): + context.append(runtime_message('vzdump.size', language, value=data['size'])) + if data.get('duration'): + context.append(runtime_message('vzdump.duration', language, value=data['duration'])) + if context: + body_text += '\n' + '\n'.join(context) + + # PVE can move a one-line setup/abort reason exclusively into its subject. + # Preserve that raw failure context, without using it as a localized title. + if event_type in ('backup_complete', 'backup_fail') and ( + event_type == 'backup_fail' or data.get('backup_outcome') == 'failed'): + source_subject = str(data.get('pve_title') or '').strip() + if source_subject and source_subject not in body_text: + body_text += '\n' + source_subject + # Clean up: collapse runs of 3+ blank lines into 1, remove trailing whitespace import re as _re body_text = _re.sub(r'\n{3,}', '\n\n', body_text.strip()) diff --git a/AppImage/scripts/tests/test_notification_runtime_i18n.py b/AppImage/scripts/tests/test_notification_runtime_i18n.py index abd1b127..6242ffae 100644 --- a/AppImage/scripts/tests/test_notification_runtime_i18n.py +++ b/AppImage/scripts/tests/test_notification_runtime_i18n.py @@ -77,9 +77,16 @@ class RuntimeCatalogTests(unittest.TestCase): "channels.email.severity.observation", "channels.email.status.unconfirmed"} for language, catalog in self.catalogs.items(): translated = flatten(catalog) - expected = set(en) - pending_slovak if language == 'sk' else set(en) - self.assertEqual(set(translated), expected, language) - for key in expected: + if language == 'sk': + # Missing maintainer-owned leaves may be generated later. + # Accept only this bounded gap, and validate every present leaf. + self.assertTrue(set(en) - pending_slovak <= set(translated), language) + self.assertTrue(set(translated) <= set(en), language) + else: + self.assertEqual(set(translated), set(en), language) + for key in translated: + self.assertIsInstance(translated[key], str, f"{language}:{key}") + self.assertTrue(translated[key].strip(), f"{language}:{key}") if language == 'sk' and key in ('templates.backup_complete.title', 'templates.backup_complete.body'): continue # exact upstream SK, superseded only at render time diff --git a/AppImage/scripts/tests/test_vzdump_ai_integrity.py b/AppImage/scripts/tests/test_vzdump_ai_integrity.py index 458606bf..2286842c 100644 --- a/AppImage/scripts/tests/test_vzdump_ai_integrity.py +++ b/AppImage/scripts/tests/test_vzdump_ai_integrity.py @@ -275,8 +275,14 @@ class VzdumpAIIntegrityTests(unittest.TestCase): rendered["body"], "CRITICAL", data, ) + # The failure title now identifies the unique failed guest; inventory + # remains exactly once in the detail table, not suppressed from body. + inventory = html.split('', 1)[1].split('
', 1)[0] for vmid in range(100, 149): - self.assertEqual(html.count(f"guest-{vmid} ({vmid})"), 1, vmid) + self.assertEqual(inventory.count(f"guest-{vmid} ({vmid})"), 1, vmid) + self.assertIn('guest-148 (148)', rendered['title']) + self.assertNotIn('guest-100 (100)', rendered['title']) + self.assertEqual(html.count('guest-148 (148)'), 2) # title + inventory self.assertEqual(html.count("49 backups"), 1) self.assertEqual(html.count("1 failed"), 1) self.assertEqual(html.count(">Zlyhalo<"), 1) From 147c40186445509cbb3238d4260e8647e3f19f16 Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Wed, 30 Sep 2026 00:37:28 +0200 Subject: [PATCH 07/14] fix(notifications): retain diagnostic context across final consumers --- .../tests/notification_final_fixture.py | 88 +++++++++++ .../test_notification_final_corrections.py | 146 ++++++++++++++++++ AppImage/scripts/notification_channels.py | 21 ++- AppImage/scripts/notification_manager.py | 30 +++- AppImage/scripts/notification_templates.py | 13 +- .../tests/test_notification_runtime_i18n.py | 5 +- 6 files changed, 287 insertions(+), 16 deletions(-) create mode 100644 .github/scripts/tests/notification_final_fixture.py create mode 100644 .github/scripts/tests/test_notification_final_corrections.py diff --git a/.github/scripts/tests/notification_final_fixture.py b/.github/scripts/tests/notification_final_fixture.py new file mode 100644 index 00000000..a6e9b17d --- /dev/null +++ b/.github/scripts/tests/notification_final_fixture.py @@ -0,0 +1,88 @@ +"""Inert infrastructure for actual endpoint/manual/queued/SQLite release seams.""" +import datetime +import sqlite3 +import tempfile +import threading +import time +import types +import typing +from html.parser import HTMLParser +from pathlib import Path +from notification_fixture import templates, SCRIPTS, EmailChannel, extract + + +def visible(markup): + class Text(HTMLParser): + def __init__(self): super().__init__(); self.parts = []; self.tags = [] + def handle_data(self, data): self.parts.append(data) + def handle_starttag(self, tag, attrs): self.tags.append(tag) + parser = Text(); parser.feed(markup) + return '\n'.join(p.strip() for p in parser.parts if p.strip()), parser.tags + + +def restore_event(warnings): + events = [] + ns = {'request': types.SimpleNamespace(remote_addr='127.0.0.1', get_json=lambda **kw: { + 'hostname':'node-a', 'guests':'3', 'stubs':'1', 'stale_nodes':'0', + 'components':'2', 'duration':'2m', 'warnings':warnings}), + 'notification_manager':types.SimpleNamespace(emit_event=lambda **kw:events.append(kw)), + 'jsonify':lambda value:value} + handler = extract(SCRIPTS/'flask_notification_routes.py', 'internal_restore_event', None, ns) + response, status = handler() + assert status == 200 and len(events) == 1 + return events[0] + + +def deliver(event_type, data, severity='INFO', language='en', manual=False, quiet=False, quiet_before=()): + """Execute real dispatch/manual and optional real SQLite buffer+flush.""" + captured = [] + channel = object.__new__(EmailChannel); channel.subject_prefix = '[ProxMenux]' + def sink(title, body, severity, data): + markup = channel._format_html(title, body, severity, data) + text, tags = visible(markup) + captured.append(dict(title=title, body=body, severity=severity, data=dict(data), html=markup, text=text, tags=tags)) + return {'success':True} + manager = types.SimpleNamespace(_config={'email.rich_format':'true'}, _lock=threading.RLock(), + _channels={'email':types.SimpleNamespace(send=sink)}, + _group_limiter=types.SimpleNamespace(allow=lambda group:True), + _claim_delivery=lambda event:'inert', _finish_delivery_claim=lambda *a,**kw:None, + _notification_language=lambda:language, _build_ai_config=lambda:{'ai_enabled':'false'}, + _in_quiet_hours=lambda channel:quiet, _should_buffer_for_digest=lambda *a:False, + _record_history=lambda *a:None, _stats={'total_sent':0,'total_errors':0}, + is_event_enabled=lambda event:True) + ns = dict(vars(typing), NotificationEvent=types.SimpleNamespace, TEMPLATES=templates.TEMPLATES, render_template=templates.render_template, + resolve_notification_hostname=lambda host, config:host or 'node-a', + enrich_with_emojis=templates.enrich_with_emojis, datetime=datetime.datetime, + _should_bypass_ai=lambda event:True, _AI_BYPASS_EVENTS=frozenset({'backup_complete','backup_fail'})) + for name in ('_dispatch_to_channels', '_dispatch_event', 'send_notification'): + setattr(manager, name, types.MethodType(extract(SCRIPTS/'notification_manager.py', name, 'NotificationManager', ns), manager)) + with tempfile.TemporaryDirectory(prefix='notification-final-') as scratch: + db = Path(scratch)/'pending.sqlite' + rows = [] + if quiet: + conn = sqlite3.connect(db) + conn.execute('CREATE TABLE quiet_pending (id INTEGER PRIMARY KEY, channel TEXT, event_type TEXT, event_group TEXT, severity TEXT, ts INTEGER, title TEXT, body TEXT)') + conn.commit(); conn.close() + qns = dict(vars(typing), sqlite3=sqlite3, DB_PATH=db, time=time, datetime=datetime.datetime, + _resolve_display_hostname=lambda config:'node-a', runtime_message=templates.runtime_message, + EVENT_EMOJI=templates.EVENT_EMOJI, CATEGORY_EMOJI=templates.CATEGORY_EMOJI) + for name in ('_buffer_quiet_event', '_flush_quiet_for_channel', '_compose_digest_body'): + setattr(manager,name,types.MethodType(extract(SCRIPTS/'notification_manager.py',name,'NotificationManager',qns),manager)) + for earlier in quiet_before: + manager._dispatch_event(types.SimpleNamespace(**earlier,source='inert',entity_type='node',entity_id='',event_id='inert',fingerprint='inert')) + if manual: + assert not quiet + result = manager.send_notification(event_type, severity, '', '', dict(data)) + assert result['success'] + else: + event = types.SimpleNamespace(event_type=event_type, severity=severity, data=dict(data), source='inert', entity_type='node', entity_id='', event_id='inert', fingerprint='inert') + manager._dispatch_event(event) + if quiet: + conn = sqlite3.connect(db); rows = conn.execute('SELECT event_type,severity,title,body FROM quiet_pending').fetchall(); conn.close() + assert len(rows)==1+len(quiet_before) and not captured + manager._flush_quiet_for_channel('email', manager._channels['email']) + conn=sqlite3.connect(db); remaining=conn.execute('SELECT count(*) FROM quiet_pending').fetchone()[0]; conn.close() + assert remaining==0 + assert len(captured)==1 + if quiet: captured[0]['buffered']=rows + return captured[0] diff --git a/.github/scripts/tests/test_notification_final_corrections.py b/.github/scripts/tests/test_notification_final_corrections.py new file mode 100644 index 00000000..2976991a --- /dev/null +++ b/.github/scripts/tests/test_notification_final_corrections.py @@ -0,0 +1,146 @@ +"""Final review contracts at actual locale/manual/queued/quiet email seams. +All operational dependencies are inert; no manager or route module import. +""" +import ast +import copy +import html +import unittest +from unittest.mock import patch +from notification_fixture import templates, SCRIPTS, LANGUAGES, receive +from notification_final_fixture import deliver, restore_event + +# Literal native send_notification body, produced by pinned PVE Perl helpers. +NATIVE_REPORT = 'Details\n=======\n' + '''VMID Name Status Time Size Filename +100 web ok 1m 1s 1 GiB vm/100/2026-09-29T17:00:00Z + +Total running time: 1m 1s +Total size: 1 GiB + +Logs +==== +vzdump --all 1 --storage PBS --mode snapshot + +100: no log available +''' + + + +class FinalCorrectionsTests(unittest.TestCase): + def test_runtime_backup_assertions_accept_generated_and_missing_slovak_title(self): + path = SCRIPTS / 'tests/test_notification_runtime_i18n.py' + tree = ast.parse(path.read_text()) + owner = next(n for n in tree.body if isinstance(n, ast.ClassDef) and n.name == 'RuntimeCatalogTests') + method = next(n for n in owner.body if isinstance(n, ast.FunctionDef) and n.name == 'test_special_formatters_digest_and_test_message_are_slovak') + start = next(i for i,n in enumerate(method.body) if isinstance(n,ast.Assign) and any(isinstance(t,ast.Name) and t.id=='backup' for t in n.targets)) + stop = next(i for i,n in enumerate(method.body) if isinstance(n,ast.Assign) and any(isinstance(t,ast.Name) and t.id=='manager' for t in n.targets)) + block = ast.Module(body=method.body[start:stop], type_ignores=[]) + english = templates._load_runtime_catalog('en') + for title in (None, '{hostname}: GENERATED_SK potvrdené'): + sk = copy.deepcopy(templates._load_runtime_catalog('sk')) + if title is not None: sk.setdefault('backup', {})['confirmedTitle'] = title + else: sk.get('backup', {}).pop('confirmedTitle', None) + with patch.object(templates, '_load_runtime_catalog', side_effect=lambda lang: english if lang == 'en' else sk): + exec(compile(block,str(path),'exec'), {'self':self, 'notification_templates':templates}) + + + def test_native_multiline_failure_diagnostics_survive_receiver_dispatch_once(self): + subject = 'vzdump backup status (node-a): backup failed: multiple problems' + for diagnostic, report in ( + ('external provider job cleanup failed\njob-abort hook permission denied', NATIVE_REPORT), + ('job interrupted\njob-abort hook permission denied', NATIVE_REPORT.replace('ok ', 'todo ')), + ('unable to initialize external provider\njob-abort hook permission denied', NATIVE_REPORT.replace('100 web ok 1m 1s 1 GiB vm/100/2026-09-29T17:00:00Z\n', '')), + ): + event = receive(diagnostic + '\n' + report, 'error', subject) + self.assertEqual((event.event_type, event.severity), ('backup_fail', 'CRITICAL')) + for language in LANGUAGES: + result = deliver(event.event_type, event.data, event.severity, language) + for line in diagnostic.splitlines(): + self.assertEqual(result['body'].count(line), 1) + self.assertEqual(result['text'].count(line), 1) + if 'ok ' in report: + self.assertIn('✅ VM web (100)', result['body']) + + + def test_restore_endpoint_quiet_release_keeps_warning_counts_and_truthful_footer(self): + reason = 'missing module zfs & {literal}' + event = restore_event(reason) + self.assertEqual(event['severity'], 'WARNING') + for language in LANGUAGES: + result = deliver(event['event_type'], event['data'], event['severity'], language, quiet=True) + self.assertEqual(result['severity'], 'INFO') + buffered_body = result['buffered'][0][3] + self.assertEqual(result['text'].count(reason), 1) + for line in buffered_body.splitlines(): + if line.strip(): self.assertIn(line.strip(), result['text']) + self.assertIn('2m', result['text']) + self.assertNotIn(templates.runtime_message('digest.footer', language), result['body']) + self.assertNotIn('script', result['tags']) + + + def test_long_disappearance_reason_is_present_once_in_actual_dispatch(self): + reason = 'Temperature exceeded configured limit; the source stopped reporting this observation after expiry.' + for language in LANGUAGES: + for manual in (False, True): + result = deliver('error_resolved', {'hostname':'node-a','category':'temperature', + 'reason':reason,'duration':'3d 2h','original_severity':'WARNING'}, 'OK', language, manual=manual) + self.assertEqual(result['text'].count(reason), 1) + self.assertNotIn('>OK', result['html']) + self.assertNotIn('>RESOLVED', result['html']) + + + def test_raw_restore_and_observation_cells_use_event_scoped_mail_wrapping(self): + token = 'b' * 64 + event = restore_event('Boot check: recorded token ' + token + '; verification pending') + for language in LANGUAGES: + results = [deliver(event['event_type'],event['data'],event['severity'],language,quiet=quiet) for quiet in (False,True)] + results.append(deliver('error_resolved',{'hostname':'node-a','reason':token,'category':'temperature','duration':'3d 2h'},'OK',language)) + for result in results: + self.assertIn('table-layout:fixed;', result['html']) + self.assertIn('word-wrap:break-word;', result['html']) + self.assertIn('overflow-wrap:break-word;', result['html']) + self.assertEqual(result['text'].count(token), 1) + unrelated = deliver('node_reconnect', {'hostname':'node-a'}, 'OK') + self.assertNotIn('table-layout:fixed;', unrelated['html']) + + + def test_manual_failed_and_unconfirmed_backup_keep_short_actionable_reason(self): + for outcome, reason in (('failed','PBS permission denied for datastore remote'), + ('unconfirmed','Task status unavailable: upstream API timed out')): + for language in LANGUAGES: + result = deliver('backup_complete',{'hostname':'node-a','vmid':'100','vmname':'web', + 'storage':'PBS','backup_outcome':outcome,'reason':reason},'INFO',language,manual=True) + self.assertEqual(result['text'].count(reason), 1) + self.assertNotIn('>COMPLETED', result['html']) + + + def test_backup_reason_threshold_and_raw_body_deduplication(self): + for event in ('backup_complete', 'backup_fail'): + for length in (79, 80, 81, 120): + prefix = ' & {rack.location} ' + reason = prefix + 'b' * (length - len(prefix)) + self.assertEqual(len(reason), length) + for raw in ('', reason): + data = {'hostname':'node-a','backup_outcome':'failed','reason':reason,'pve_message':raw} + for manual in (False, True): + result = deliver(event,data,'INFO','en',manual=manual) + self.assertEqual(result['text'].count(reason), 1, (event,length,raw,manual)) + self.assertNotIn('raw', result['tags']) + + + def test_quiet_restore_after_preview_limit_is_not_omitted_or_called_info(self): + event = restore_event('missing module zfs') + earlier = [dict(event_type='service_fail',severity='WARNING',data={'hostname':'node-a','service_name':f'unit-{i}','reason':'recorded'}) for i in range(8)] + result = deliver(event['event_type'],event['data'],event['severity'],quiet=True,quiet_before=earlier) + self.assertEqual(result['text'].count('missing module zfs'), 1) + self.assertNotIn(templates.runtime_message('digest.lead','en',count=9), result['body']) + self.assertEqual(result['data']['_count'], 9) + + + def test_native_error_block_does_not_deduplicate_a_substring_of_inventory(self): + event = receive('web\nexternal provider job cleanup failed\n' + NATIVE_REPORT, + 'error','vzdump backup status (node-a): backup failed: multiple problems') + result = deliver(event.event_type,event.data,event.severity) + self.assertEqual(result['body'].splitlines().count('web'), 1) + + +if __name__ == '__main__': unittest.main() diff --git a/AppImage/scripts/notification_channels.py b/AppImage/scripts/notification_channels.py index eddc2bde..bf74ddeb 100644 --- a/AppImage/scripts/notification_channels.py +++ b/AppImage/scripts/notification_channels.py @@ -1053,11 +1053,13 @@ class EmailChannel(NotificationChannel): status = 'unconfirmed' sev['label'] = _runtime_text(f'email.status.{status}', data) group = data.get('_group', 'other') - # Scoped inline mail-compatible wrapping: temperature measurements - # and backup identities/raw diagnostics. Other events retain layout. + # Scoped inline mail-compatible wrapping for authoritative raw-context + # bodies, including restore bodies released from quiet hours. backup_email = event_type in {'backup_complete', 'backup_fail'} - temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if event_type == 'temp_high' or backup_email else '' - temp_table_layout = 'table-layout:fixed;' if event_type == 'temp_high' or backup_email else '' + wrap_body = (event_type in {'temp_high', 'system_restore_completed', 'error_resolved'} + or backup_email or data.get('_restore_summary')) + temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else '' + temp_table_layout = 'table-layout:fixed;' if wrap_body else '' backup_title_wrap = temp_cell_wrap if backup_email else '' backup_metadata_layout = 'table-layout:fixed;' if backup_email else '' section_label = _runtime_text(f'email.groups.{group}', data) @@ -1090,8 +1092,14 @@ class EmailChannel(NotificationChannel): ('', html_mod.escape(line.strip())) for line in body.split('\n') if line.strip() ) + # A metadata-only/manual body may be generic. Keep actionable raw + # context once, without restoring duplicated inventory metadata. + reason = data.get('reason', '') + if reason and len(reason) <= 80 and reason not in body: + detail_rows.append((html_mod.escape(_runtime_text('email.fields.reason', data)), + html_mod.escape(reason))) - if event_type in {'system_restore_completed', 'error_resolved'}: + if event_type in {'system_restore_completed', 'error_resolved'} or data.get('_restore_summary'): # Observation age/disappearance must not become a green OK row. # The endpoint's warnings_block and task counts live in the # localized body, not the generic services Event row. @@ -1129,7 +1137,8 @@ class EmailChannel(NotificationChannel): # ── Reason / details block (long text, displayed separately) ── reason = data.get('reason', '') reason_html = '' - if reason and len(reason) > 80 and not (event_type == 'temp_high' and reason in body): + if reason and len(reason) > 80 and not ( + (event_type in {'temp_high', 'error_resolved'} or backup_email) and reason in body): reason_html = f'''

{_runtime_text('email.details', data)}

diff --git a/AppImage/scripts/notification_manager.py b/AppImage/scripts/notification_manager.py index 332072a9..7f317314 100644 --- a/AppImage/scripts/notification_manager.py +++ b/AppImage/scripts/notification_manager.py @@ -1821,7 +1821,8 @@ class NotificationManager: print(f"[NotificationManager] digest cleanup failed for " f"{ch_name}: {e}") - def _compose_digest_body(self, rows: list, use_icons: bool = False) -> str: + def _compose_digest_body(self, rows: list, use_icons: bool = False, + quiet_release: bool = False) -> str: """Render a grouped summary body. rows is a list of (id, event_type, event_group, ts, title, body) tuples ordered by timestamp ASC. @@ -1830,16 +1831,23 @@ class NotificationManager: groups: OrderedDict[str, list] = OrderedDict() for _id, ev_type, group, ts, title, body in rows: label = group or 'other' - groups.setdefault(label, []).append((ts, ev_type, title)) + groups.setdefault(label, []).append((ts, ev_type, title, body)) language = self._notification_language() - lines = [runtime_message('digest.lead', language, count=len(rows))] + # The quiet summary title already carries the total; the daily lead + # incorrectly calls every buffered WARNING an INFO event. + lines = [] if quiet_release else [runtime_message('digest.lead', language, count=len(rows))] for group, items in groups.items(): group_label = runtime_message(f'digest.groups.{group}', language) or group.title() group_icon = CATEGORY_EMOJI.get(group, '') if use_icons else '' group_prefix = f'{group_icon} ' if group_icon else '' lines.append(f"{group_prefix}{group_label}: {len(items)}") - for ts, ev_type, title in items[:8]: + # Quiet hours can buffer restore warnings, unlike the daily INFO + # digest. Keep their complete recorded body, even past the usual + # title preview limit, without changing either delivery policy. + visible_items = [item for index, item in enumerate(items) + if index < 8 or (quiet_release and item[1] == 'system_restore_completed')] + for ts, ev_type, title, body in visible_items: hhmm = datetime.fromtimestamp(ts).strftime('%H:%M') short_title = title.split(': ', 1)[-1] if ': ' in title else title event_icon = ( @@ -1847,10 +1855,15 @@ class NotificationManager: ) if use_icons else '' event_prefix = f'{event_icon} ' if event_icon else '' lines.append(f" • {event_prefix}{hhmm} {short_title}") - if len(items) > 8: - lines.append(runtime_message('digest.more', language, count=len(items) - 8)) + if quiet_release and ev_type == 'system_restore_completed' and body: + lines.extend(line for line in body.splitlines() if line.strip()) + if len(items) > len(visible_items): + lines.append(runtime_message('digest.more', language, count=len(items) - len(visible_items))) lines.append('') - lines.append(runtime_message('digest.footer', language)) + # The daily footer describes live warning delivery, which is not true + # for warnings buffered during quiet hours. Do not repeat that claim. + if not quiet_release: + lines.append(runtime_message('digest.footer', language)) return '\n'.join(lines).rstrip() + '\n' # ─── Quiet Hours buffer + flush ──────────────────────────── @@ -1985,13 +1998,14 @@ class NotificationManager: 'digest.quietTitle', language, hostname=host, count=len(rows), ) use_icons = self._config.get(f'{ch_name}.rich_format', 'false') == 'true' - summary_body = self._compose_digest_body(rows, use_icons=use_icons) + summary_body = self._compose_digest_body(rows, use_icons=use_icons, quiet_release=True) result: dict = {'success': False, 'error': ''} try: result = channel.send( summary_title, summary_body, severity='INFO', data={'_quiet_hours_summary': True, '_count': len(rows), + '_restore_summary': any(row[1] == 'system_restore_completed' for row in rows), '_notification_language': language}, ) or result except Exception as e: diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index 69d6bdbd..47dbc221 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -2090,8 +2090,19 @@ def render_template(event_type: str, data: Dict[str, Any], diagnostic_lines = [line.strip() for line in pve_message.splitlines() if re.match(r'^\s*(?:\d+:\s*)?(?:\d{4}-\d{2}-\d{2}\s+\S+\s+)?(?:WARN(?:ING)?:|ERROR:|TASK ERROR)', line, re.IGNORECASE)] + if event_type == 'backup_fail' or data.get('backup_outcome') == 'failed': + # Native send_notification puts multiline job/setup errors + # before Details, while its subject says only "multiple problems". + # Keep that raw block when inventory replaces the producer body; + # it is job context, not evidence that every guest failed. + error_block = re.match(r'\A(.*?)^Details\r?\n=+\s*$', + pve_message, re.MULTILINE | re.DOTALL) + if error_block: + diagnostic_lines = error_block.group(1).rstrip('\r\n').splitlines() + diagnostic_lines if diagnostic_lines: - body_text += '\n' + '\n'.join(dict.fromkeys(diagnostic_lines)) + body_text += '\n' + '\n'.join( + line for line in dict.fromkeys(diagnostic_lines) + if line.strip() and line not in body_text.splitlines()) else: # Couldn't parse -- use PVE raw message as body body_text = pve_message.strip() diff --git a/AppImage/scripts/tests/test_notification_runtime_i18n.py b/AppImage/scripts/tests/test_notification_runtime_i18n.py index 6242ffae..6ac97b01 100644 --- a/AppImage/scripts/tests/test_notification_runtime_i18n.py +++ b/AppImage/scripts/tests/test_notification_runtime_i18n.py @@ -341,7 +341,10 @@ class RuntimeCatalogTests(unittest.TestCase): }, language="sk", ) - self.assertIn("Backup complete", backup["title"]) + expected_title = notification_templates.runtime_message( + "backup.confirmedTitle", "sk", hostname="pve01", + ) + self.assertTrue(backup["title"].startswith(expected_title + " — ")) self.assertIn("pbs-main", backup["title"]) self.assertIn("VM alpha (100)", backup["title"]) self.assertNotIn("Backup job finished", backup["title"]) From 08598d2863f8afaa5b6cfc0240c55f952f23a41f Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:48:37 +0200 Subject: [PATCH 08/14] fix(notifications): preserve evidenced outcomes in backup and health notices --- .../tests/test_command_descriptions.py | 4 + .../tests/test_notification_corrections.py | 4 +- .../test_notification_maintainer_followup.py | 150 ++++++++++++++++++ .../test_notification_outcome_wording.py | 8 +- .../scripts/tests/test_notification_pve92.py | 4 +- .../test_notification_recovery_evidence.py | 107 +++++++++++++ AppImage/messages/de/common.json | 9 +- AppImage/messages/en/common.json | 2 +- AppImage/messages/es/common.json | 9 +- AppImage/messages/fr/common.json | 9 +- AppImage/messages/it/common.json | 9 +- AppImage/messages/pt/common.json | 9 +- AppImage/messages/sv/common.json | 9 +- AppImage/scripts/health_monitor.py | 6 +- AppImage/scripts/health_persistence.py | 53 ++++++- AppImage/scripts/notification_channels.py | 22 ++- AppImage/scripts/notification_events.py | 26 ++- AppImage/scripts/notification_manager.py | 23 ++- AppImage/scripts/notification_templates.py | 80 ++++++++-- .../tests/test_notification_runtime_i18n.py | 5 +- 20 files changed, 497 insertions(+), 51 deletions(-) create mode 100644 .github/scripts/tests/test_notification_maintainer_followup.py create mode 100644 .github/scripts/tests/test_notification_recovery_evidence.py diff --git a/.github/scripts/tests/test_command_descriptions.py b/.github/scripts/tests/test_command_descriptions.py index be4ad419..16e0e9a2 100644 --- a/.github/scripts/tests/test_command_descriptions.py +++ b/.github/scripts/tests/test_command_descriptions.py @@ -142,6 +142,10 @@ class CommandDescriptionsTests(unittest.TestCase): 'observation', source['channels']['email']['severity']['observation']) local['channels']['email']['status'].setdefault( 'unconfirmed', source['channels']['email']['status']['unconfirmed']) + local['channels']['email']['status'].setdefault( + 'completed_with_warnings', source['channels']['email']['status']['completed_with_warnings']) + for key, value in source['healthRecovery'].items(): + local.setdefault('healthRecovery', {}).setdefault(key, value) path.write_text(json.dumps(temporary, ensure_ascii=False)) # Model steady state after the bot fills these intentional new # messages; keep repository locales and all other leaves intact. diff --git a/.github/scripts/tests/test_notification_corrections.py b/.github/scripts/tests/test_notification_corrections.py index 94ae1469..c9bfaa22 100644 --- a/.github/scripts/tests/test_notification_corrections.py +++ b/.github/scripts/tests/test_notification_corrections.py @@ -217,10 +217,10 @@ class CorrectionTests(unittest.TestCase): self.assertIn('✅ VM web (100)', result['body']) - def test_official_prefixed_warning_is_uncertain_and_retained(self): + def test_official_completed_report_warning_is_distinct_and_retained(self): warning = '100: 2026-09-29 17:00:00 WARN: unable to add notes - permission denied' event = receive(REPORT + '\n' + warning) - self.assertEqual(event.data['backup_outcome'], 'unconfirmed') + self.assertEqual(event.data['backup_outcome'], 'completed_with_warnings') for language in LANGUAGES: result, markup = email(event.event_type, event.data, event.severity, language) self.assertIn(warning, result['body']) diff --git a/.github/scripts/tests/test_notification_maintainer_followup.py b/.github/scripts/tests/test_notification_maintainer_followup.py new file mode 100644 index 00000000..42d37377 --- /dev/null +++ b/.github/scripts/tests/test_notification_maintainer_followup.py @@ -0,0 +1,150 @@ +"""Maintainer acceptance at inert actual notification consumers.""" +import unittest +from notification_fixture import templates, receive, LANGUAGES, SCRIPTS, extract +from notification_final_fixture import deliver + +NATIVE_REPORT = '''Details +======= +VMID Name Status Time Size Filename +100 web ok 1m 1s 1 GiB vm/100/2026-09-29T17:00:00Z + +Total running time: 1m 1s +Total size: 1 GiB + +Logs +==== +vzdump --all 1 --storage PBS --mode snapshot + +100: 2026-09-29 17:00:00 INFO: Starting Backup of VM 100 (qemu) +100: 2026-09-29 17:01:01 INFO: Finished Backup of VM 100 (00:01:01) +''' + +class MaintainerFollowupTests(unittest.TestCase): + def test_completed_with_warnings_requires_independent_completion(self): + warning = '\n100: 2026-09-29 17:00:01 WARN: file changed during backup' + event = receive(NATIVE_REPORT + warning) + self.assertEqual(event.data['backup_outcome'], 'completed_with_warnings') + self.assertEqual((event.event_type,event.severity), ('backup_complete','INFO')) + for manual in (False,True): + for lang in LANGUAGES: + result = deliver(event.event_type,event.data,event.severity,lang,manual=manual) + label = templates.runtime_message('backup.warningTitle',lang,hostname='node-a') + self.assertIn(label,result['title']) + status = templates.runtime_message('channels.email.status.completed_with_warnings',lang) + self.assertIn(status,result['text']) + self.assertIn('file changed during backup',result['text']) + # Manual sends intentionally skip channel emoji enrichment. + if not manual: self.assertTrue(result['title'].startswith('💾⚠️')) + rich, _ = templates.enrich_with_emojis(event.event_type,result['title'],result['body'],event.data) + self.assertTrue(rich.startswith('💾⚠️')) + self.assertEqual(receive('WARN: file changed during backup').data['backup_outcome'],'unconfirmed') + self.assertEqual(receive(NATIVE_REPORT.split('Total running time:')[0]+warning).data['backup_outcome'],'unconfirmed') + self.assertEqual(receive(NATIVE_REPORT+warning+'\nERROR: cleanup failed').data['backup_outcome'],'failed') + self.assertEqual(receive(NATIVE_REPORT+warning,'warning').data['backup_outcome'],'completed_with_warnings') + self.assertEqual(receive('INFO: Starting Backup of VM 100 (qemu)\nINFO: Finished Backup of VM 100 (00:01:01)'+warning).data['backup_outcome'],'completed_with_warnings') + self.assertEqual(receive(NATIVE_REPORT+warning,kind='').data['backup_outcome'],'unconfirmed') + + def test_null_filename_failed_guest_uses_own_start_identity(self): + report = NATIVE_REPORT.replace('100 web ok 1m 1s 1 GiB vm/100/2026-09-29T17:00:00Z', + '100 web err 1m 1s 0 B null') + for kind, prefix in (('qemu','VM'),('lxc','CT')): + own = report.replace('VM 100 (qemu)',f'VM 100 ({kind})') + # Unrelated guest appears first and must never supply the failed type. + message = f'INFO: Starting Backup of VM 999 ({"lxc" if kind == "qemu" else "qemu"})\n' + own + event = receive(message,'error','vzdump backup status (raw-node): backup failed') + parsed = templates._parse_vzdump_message(message) + self.assertEqual(parsed['vms'][0]['type'],kind) + result = deliver(event.event_type,event.data,event.severity) + self.assertIn(f'{prefix} web (100)',result['title']) + self.assertIn(f'❌ {prefix} web (100)',result['body']) + for message in (report.replace('Starting Backup of VM 100','Starting Backup of VM 999'), + report+'\nINFO: Starting Backup of VM 100 (lxc)'): + self.assertEqual(templates._parse_vzdump_message(message)['vms'][0]['type'],'') + + def test_original_subject_only_retained_when_no_guest_context(self): + subject = 'vzdump backup status (raw-host): backup failed: multiple problems' + event = receive('ERROR: archive write failed\n'+NATIVE_REPORT,'error',subject) + event.data['hostname']='configured-alias' + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang) + self.assertNotIn(subject,result['text']) + self.assertEqual(result['text'].count('ERROR: archive write failed'),1) + self.assertIn('configured-alias',result['title']) + setup=receive('Details\n=======\nVMID Name Status Time Size Filename\n\nTotal running time: 0s\nTotal size: 0 B','error',subject.replace('multiple problems','unable to open storage')) + result=deliver(setup.event_type,setup.data,setup.severity) + self.assertEqual(result['text'].count(setup.data['pve_title']),1) + unique=receive(NATIVE_REPORT,'error',subject.replace('multiple problems','job-end hook denied')) + result=deliver(unique.event_type,unique.data,unique.severity) + self.assertIn('job-end hook denied',result['body']) + self.assertNotIn('vzdump backup status',result['body']) + + def test_backup_diagnostics_are_bounded_with_principal_cause_and_notice(self): + raw = NATIVE_REPORT + '\n' + '\n'.join(f'WARN: repeated warning {i}' for i in range(80)) + '\nERROR: principal archive write failure' + event=receive(raw) + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang) + self.assertIn('ERROR: principal archive write failure',result['body']) + self.assertLessEqual(len('\n'.join(line for line in result['body'].splitlines() if line.startswith(('WARN:','ERROR:')))),1024) + self.assertLessEqual(sum(line.startswith(('WARN:','ERROR:')) for line in result['body'].splitlines()),8) + notice=templates.runtime_message('backup.diagnosticsOmitted',lang,count=73) + self.assertIn(notice,result['body']) + self.assertEqual(event.data['pve_message'],raw) + long=receive('ERROR: '+ 'b'*5000,'error','vzdump backup status (node): backup failed') + result=deliver(long.event_type,long.data,long.severity) + self.assertLess(len(result['body']),1400) + self.assertIn('ERROR: '+ 'b'*100,result['body']) + self.assertIn(templates.runtime_message('backup.diagnosticsOmitted','en',count=1),result['body']) + + def test_real_quiet_digest_retains_each_backup_outcome_icon(self): + samples=[(NATIVE_REPORT,'confirmed','💾✅'),(NATIVE_REPORT+'\nWARN: changed file','completed_with_warnings','💾⚠️'),(NATIVE_REPORT+'\nERROR: write failed','failed','💾❌'),('INFO: Starting Backup of VM 100 (qemu)','unconfirmed','💾❔')] + for raw,outcome,icon in samples: + event=receive(raw) + self.assertEqual(event.data['backup_outcome'],outcome) + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang,quiet=True) + self.assertIn(icon,result['body']) + self.assertNotIn('💾❔',result['body']) if outcome!='unconfirmed' else None + import types,datetime + ns={'datetime':datetime.datetime,'runtime_message':templates.runtime_message,'EVENT_EMOJI':templates.EVENT_EMOJI,'CATEGORY_EMOJI':templates.CATEGORY_EMOJI} + compose=extract(SCRIPTS/'notification_manager.py','_compose_digest_body','NotificationManager',ns) + target=types.SimpleNamespace(_notification_language=lambda:'en') + rows=[(i,'backup_complete','backup',1,icon+' node: Backup','') for i,(_,_,icon) in enumerate(samples)] + body=compose(target,rows,use_icons=True) + for _,_,icon in samples:self.assertIn(icon,body) + plain=compose(target,rows,use_icons=False) + for _,_,icon in samples:self.assertNotIn(icon,plain) + should=extract(SCRIPTS/'notification_manager.py','_should_buffer_for_digest','NotificationManager',{}) + self.assertFalse(should(types.SimpleNamespace(_DIGEST_EXEMPT_EVENTS={'backup_complete'},_config={'email.digest_enabled':'true'}),'email','INFO','backup_complete')) + + def test_quiet_restore_details_are_subordinate_in_text_and_email(self): + from notification_final_fixture import restore_event + event=restore_event('missing module zfs') + for lang in LANGUAGES: + result=deliver(event['event_type'],event['data'],event['severity'],lang,quiet=True) + body_lines=result['buffered'][0][3].splitlines() + for line in body_lines: + if line.strip(): + self.assertIn(' '+line.strip(),result['body']) + self.assertIn(' '+__import__('html').escape(line.strip()),result['html']) + self.assertIn('white-space:pre-wrap;',result['html']) + self.assertNotIn(templates.runtime_message('digest.footer',lang),result['body']) + + def test_job_level_subject_cause_survives_warning_cap(self): + raw=NATIVE_REPORT+'\n'+'\n'.join('WARN: repeated diagnostic '+str(i) for i in range(80)) + event=receive(raw,'error','vzdump backup status (raw-host): backup failed: job-end hook denied') + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang) + self.assertEqual(result['body'].count('job-end hook denied'),1) + self.assertNotIn('vzdump backup status',result['body']) + self.assertIn(templates.runtime_message('backup.diagnosticsOmitted',lang,count=73),result['body']) + + def test_backup_quiet_email_uses_event_scoped_wrapping(self): + event=receive(NATIVE_REPORT+'\nWARN: changed file') + result=deliver(event.event_type,event.data,event.severity,'sv',quiet=True) + self.assertTrue(result['data'].get('_backup_summary')) + self.assertIn('table-layout:fixed;',result['html']) + self.assertIn('overflow-wrap:break-word;',result['html']) + unrelated=deliver('node_reconnect',{'hostname':'node-a'},'OK') + self.assertNotIn('table-layout:fixed;',unrelated['html']) + +if __name__ == '__main__': unittest.main() diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py index 5be667c3..9dabb21f 100644 --- a/.github/scripts/tests/test_notification_outcome_wording.py +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -134,7 +134,7 @@ class OutcomeWording(unittest.TestCase): ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\n'+header+'\n'+row_ok+'\nTotal running time: 00:01:00\n'+truncated, 'confirmed'), ('vzdump', 'info', header+'\n'+row_ok+'\nTotal running time: 00:01:00\n'+truncated, 'confirmed'), ('vzdump', 'info', header+'\n'+row_err+'\nTotal running time: 00:01:00\n'+truncated, 'failed'), - ('vzdump', 'warning', header+'\n'+row_ok+'\nTotal running time: 00:01:00', 'unconfirmed'), + ('vzdump', 'warning', header+'\n'+row_ok+'\nTotal running time: 00:01:00', 'completed_with_warnings'), ('vzdump', 'info', header+'\n'+row_ok, 'unconfirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\n'+header+'\n'+row_ok, 'unconfirmed'), ('vzdump', 'warning', header+'\n'+row_err+'\nTotal running time: 00:01:00', 'failed'), @@ -147,11 +147,11 @@ class OutcomeWording(unittest.TestCase): ('vzdump', 'info', header+'\n'+row_ok+'\n'+row_warning+'\nTotal running time: 00:02:00', 'unconfirmed'), ('vzdump', 'info', header+'\n'+row_ok[:30], 'unconfirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'confirmed'), - ('vzdump', 'warning', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'unconfirmed'), + ('vzdump', 'warning', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)', 'completed_with_warnings'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)', 'unconfirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Starting Backup of VM 105 (lxc)\nINFO: Finished Backup of VM 105 (00:01:00)', 'unconfirmed'), - ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nWARNING: skipped file', 'unconfirmed'), - ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: TASK OK\n104 alpha WARNINGS: 1', 'unconfirmed'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nWARNING: skipped file', 'completed_with_warnings'), + ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: TASK OK\n104 alpha WARNINGS: 1', 'completed_with_warnings'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Finished Backup of VM 105 (00:01:00)', 'unconfirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nINFO: Starting Backup of VM 105 (lxc)\nINFO: Finished Backup of VM 104 (00:01:00)\nINFO: Finished Backup of VM 105 (00:01:00)', 'confirmed'), ('vzdump', 'info', 'INFO: Starting Backup of VM 104 (qemu)\nERROR: backup failed for VM 104', 'failed'), diff --git a/.github/scripts/tests/test_notification_pve92.py b/.github/scripts/tests/test_notification_pve92.py index 16f27487..220d129a 100644 --- a/.github/scripts/tests/test_notification_pve92.py +++ b/.github/scripts/tests/test_notification_pve92.py @@ -128,10 +128,10 @@ class PVE92Tests(unittest.TestCase): self.assertEqual(event_for(message).data['backup_outcome'], 'confirmed') for message, severity, expected in ( (PVE92.replace('ok ', 'OK '), 'info', 'confirmed'), - (message, 'warning', 'unconfirmed'), + (message, 'warning', 'completed_with_warnings'), (message, 'error', 'failed'), (message + '\nERROR: archive write failed', 'info', 'failed'), - (message + '\nWARNING: skipped file', 'info', 'unconfirmed'), + (message + '\nWARNING: skipped file', 'info', 'completed_with_warnings'), (PVE92.replace('ok ', 'WARNINGS '), 'info', 'unconfirmed'), ): with self.subTest(message=message, severity=severity): diff --git a/.github/scripts/tests/test_notification_recovery_evidence.py b/.github/scripts/tests/test_notification_recovery_evidence.py new file mode 100644 index 00000000..45f316f8 --- /dev/null +++ b/.github/scripts/tests/test_notification_recovery_evidence.py @@ -0,0 +1,107 @@ +"""Fresh existing-check provenance; extracted consumers, real disposable SQLite.""" +import contextlib +import datetime +import json +import os +import sqlite3 +import tempfile +import time +import types +import unittest +from unittest.mock import patch +from notification_fixture import extract, SCRIPTS, templates, LANGUAGES +from notification_final_fixture import deliver + +class RecoveryEvidenceTests(unittest.TestCase): + def test_cpu_success_provenance_is_persisted_only_after_normal_samples(self): + events=[] + with tempfile.TemporaryDirectory() as scratch: + db=scratch+'/health.sqlite' + conn=sqlite3.connect(db) + conn.execute('CREATE TABLE errors(id INTEGER PRIMARY KEY,error_key TEXT,details TEXT,resolved_at TEXT,resolution_type TEXT,resolution_reason TEXT)') + conn.execute("INSERT INTO errors(error_key,details) VALUES ('cpu_usage','{}')") + conn.commit();conn.close() + @contextlib.contextmanager + def connection(): + c=sqlite3.connect(db) + try:yield c + finally:c.close() + ns={'datetime':datetime.datetime,'json':json} + resolve=extract(SCRIPTS/'health_persistence.py','_resolve_error_impl','HealthPersistence',ns) + store=types.SimpleNamespace(_db_connection=connection,_entity_from_details=lambda details:'',_record_event=lambda cursor,kind,key,data:events.append(data)) + store.resolve_error=lambda key,reason,**kw:resolve(store,key,reason,**kw) + ns={'Dict':dict,'Any':object,'os':os,'time':time,'health_persistence':store,'psutil':types.SimpleNamespace(cpu_percent=lambda **kw:20,cpu_count=lambda:4)} + check=extract(SCRIPTS/'health_monitor.py','_check_cpu_with_hysteresis','HealthMonitor',ns) + target=types.SimpleNamespace(state_history={'cpu_usage':[{'value':20,'time':time.time()-i*10} for i in range(10)]},CPU_CRITICAL=95,CPU_WARNING=85,CPU_RECOVERY=75,CPU_CRITICAL_DURATION=300,CPU_WARNING_DURATION=300,CPU_RECOVERY_DURATION=120,_check_cpu_temperature=lambda:None) + result=check(target) + self.assertEqual(result['status'],'OK') + self.assertTrue(events[-1].get('check_evidence'),events) + proof=events[-1]['check_evidence'] + self.assertEqual(proof['check'],'cpu_usage') + self.assertGreaterEqual(proof['checked_at'],time.time()-5) + # Existing generic resolve callers (cleanup/exclusion) get no proof. + conn=sqlite3.connect(db);conn.execute('UPDATE errors SET resolved_at=NULL');conn.commit();conn.close() + resolve(store,'cpu_usage','No longer present') + self.assertFalse(events[-1].get('check_evidence')) + + def test_recovery_query_requires_fresh_same_incident_proof(self): + with tempfile.TemporaryDirectory() as scratch: + db=scratch+'/health.sqlite' + conn=sqlite3.connect(db) + conn.execute('CREATE TABLE errors(id INTEGER PRIMARY KEY,error_key TEXT,first_seen TEXT,last_seen TEXT,resolved_at TEXT,acknowledged INTEGER)') + conn.execute('CREATE TABLE events(id INTEGER PRIMARY KEY,event_type TEXT,error_key TEXT,timestamp TEXT,data TEXT)') + now=datetime.datetime.now(); first=(now-datetime.timedelta(minutes=10)).isoformat(); last=(now-datetime.timedelta(minutes=1)).isoformat(); resolved=now.isoformat() + proof={'check':'cpu_usage','checked_at':now.timestamp()} + conn.execute('INSERT INTO errors VALUES(1,?,?,?,?,0)',('cpu_usage',first,last,resolved)) + conn.execute('INSERT INTO events VALUES(1,?,?,?,?)',('resolved','cpu_usage',resolved,json.dumps({'check_evidence':proof}))) + conn.commit();conn.close() + @contextlib.contextmanager + def connection(**kwargs): + c=sqlite3.connect(db) + try:yield c + finally:c.close() + ns={'datetime':datetime.datetime,'json':json,'time':time} + tree=(SCRIPTS/'health_persistence.py').read_text() + query=extract(SCRIPTS/'health_persistence.py','get_recovery_evidence','HealthPersistence',ns) if 'def get_recovery_evidence(' in tree else lambda *args:None + store=types.SimpleNamespace(_db_connection=connection) + self.assertEqual(query(store,'cpu_usage',first),proof) + self.assertIsNone(query(store,'cpu_usage','different incident')) + for field,value in [('acknowledged',1),('resolved_at',None),('last_seen',(now+datetime.timedelta(seconds=1)).isoformat())]: + conn=sqlite3.connect(db);conn.execute(f'UPDATE errors SET {field}=?',(value,));conn.commit();conn.close() + self.assertIsNone(query(store,'cpu_usage',first)) + conn=sqlite3.connect(db);conn.execute('UPDATE errors SET acknowledged=0,resolved_at=?,last_seen=?',(resolved,last));conn.commit();conn.close() + for bad in (None,{'check':'cpu_usage','checked_at':now.timestamp()-7201},{'check':'storage_removed','checked_at':now.timestamp()},{'check':'cpu_usage','checked_at':float('inf')},{'check':'cpu_usage','checked_at':10**400}): + conn=sqlite3.connect(db);conn.execute('UPDATE events SET data=?',(json.dumps({'check_evidence':bad}),));conn.commit();conn.close() + self.assertIsNone(query(store,'cpu_usage',first)) + + def test_poller_and_all_consumers_distinguish_proven_recovery_from_disappearance(self): + import sys + ns={'time':time,'json':json,'Dict':dict,'NotificationEvent':lambda *a,**kw:types.SimpleNamespace(event_type=a[0],severity=a[1],data=a[2])} + poll=extract(SCRIPTS/'notification_events.py','_check_persistent_health','PollingCollector',ns) + for proof in (None,{'check':'cpu_usage','checked_at':time.time()}): + events=[] + store=types.SimpleNamespace(get_active_errors=lambda:[],is_error_acknowledged=lambda key:False,get_recovery_evidence=lambda *a:proof) + collector=types.SimpleNamespace(_hostname='node-a',_ENTITY_MAP={'cpu':('node','')},_first_poll_done=True,_known_errors={'cpu_usage':{'category':'cpu','reason':'CPU high','severity':'WARNING','first_seen':'2026-09-30T00:00:00'}},_notified_severity={'cpu_usage':'WARNING'},_last_notified={'cpu_usage':1},_queue=types.SimpleNamespace(put=events.append),_guest_storage_error_is_now_foreign=lambda *a:False,_save_known_errors_meta=lambda:None) + with patch.dict(sys.modules,{'health_persistence':types.SimpleNamespace(health_persistence=store)}):poll(collector) + self.assertEqual(len(events),1) + event=events[0] + self.assertEqual(event.data.get('recovery_outcome'),'resolved' if proof else 'no_longer_reported') + self.assertEqual(event.data['is_recovery'],bool(proof)) + for lang in LANGUAGES: + for manual in (False,True): + result=deliver(event.event_type,event.data,event.severity,lang,manual=manual) + if proof: + self.assertIn(templates.runtime_message('healthRecovery.title',lang,hostname='node-a',category='cpu',entity_suffix=''),result['title']) + self.assertIn('background:#f0fdf4;',result['html']) + else:self.assertNotIn('background:#f0fdf4;',result['html']) + self.assertEqual(result['text'].count(event.data['reason']),1) + + def test_manual_recovery_flag_alone_is_not_authoritative_evidence(self): + data={'hostname':'node-a','category':'cpu','reason':'Observation disappeared','duration':'1h','original_severity':'WARNING','recovery_outcome':'resolved'} + for lang in LANGUAGES: + for manual in (False,True): + result=deliver('error_resolved',data,'OK',lang,manual=manual) + self.assertNotIn('background:#f0fdf4;',result['html']) + self.assertNotIn(templates.runtime_message('healthRecovery.body',lang,**data),result['body']) + +if __name__=='__main__':unittest.main() diff --git a/AppImage/messages/de/common.json b/AppImage/messages/de/common.json index 0cb5b7c0..424c5aba 100644 --- a/AppImage/messages/de/common.json +++ b/AppImage/messages/de/common.json @@ -6980,7 +6980,8 @@ "failed": "Fehlgeschlagen", "completed": "Abgeschlossen", "started": "Gestartet", - "unconfirmed": "Nicht bestätigt" + "unconfirmed": "Nicht bestätigt", + "completed_with_warnings": "Mit Warnungen abgeschlossen" }, "report": "{group} Bericht", "details": "Details", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "Die hohen Messwerte erstrecken sich über {duration}." }, + "healthRecovery": {"title": "{hostname}: Behoben - {category}{entity_suffix}", "body": "Eine aktuelle Zustandsprüfung hat für {category} wieder einen normalen Zustand festgestellt.\nVorherige Beobachtung: {reason}\nVorheriger Schweregrad: {original_severity}\nZeit seit der ersten Beobachtung: {duration}", "status": "Behoben"}, "backup": { "confirmedTitle": "{hostname}: Backup abgeschlossen", "confirmedBody": "Backup erfolgreich abgeschlossen.", "errorTitle": "{hostname}: Backup-Fehler gemeldet", "errorBody": "Der Backup-Bericht enthält einen Fehler.", - "unconfirmedBody": "Das Backup-Ergebnis ist nicht bestätigt." + "unconfirmedBody": "Das Backup-Ergebnis ist nicht bestätigt.", + "warningTitle": "{hostname}: Sicherung mit Warnungen abgeschlossen", + "warningBody": "Sicherung mit Warnungen abgeschlossen.", + "diagnosticsOmitted": "Weitere Diagnosezeilen oder Text ausgelassen: {count}. Originalbericht bleibt erhalten." } } } diff --git a/AppImage/messages/en/common.json b/AppImage/messages/en/common.json index abd42a64..7d0ca28a 100644 --- a/AppImage/messages/en/common.json +++ b/AppImage/messages/en/common.json @@ -6251,5 +6251,5 @@ "cancel": "Cancel" } }, - "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: No longer reported - {category}{entity_suffix}","body":"The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname}: Backup outcome unconfirmed","body":"The backup outcome could not be confirmed from this notice.","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or run the post-install menu (option 2) → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Settings > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice","observation":"No longer reported"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"backup":{"confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed."}}} + "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: No longer reported - {category}{entity_suffix}","body":"The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname}: Backup outcome unconfirmed","body":"The backup outcome could not be confirmed from this notice.","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or run the post-install menu (option 2) → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Settings > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice","observation":"No longer reported"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed","completed_with_warnings":"Completed with warnings"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"healthRecovery":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"The {category} condition returned to normal in a fresh health check.\nPrevious observation: {reason}\nPrevious severity: {original_severity}\nTime since first observation: {duration}","status":"Resolved"},"backup":{"confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed.","warningTitle":"{hostname}: Backup completed with warnings","warningBody":"Backup completed with warnings.","diagnosticsOmitted":"Additional diagnostic lines or text omitted: {count}. Original report retained."}}} } diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index e25e0a1a..961a228e 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -6980,7 +6980,8 @@ "failed": "Fallido", "completed": "Completado", "started": "Iniciado", - "unconfirmed": "Sin confirmar" + "unconfirmed": "Sin confirmar", + "completed_with_warnings": "Completado con advertencias" }, "report": "Informe de {group}", "details": "Detalles", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "Lecturas altas registradas a lo largo de {duration}." }, + "healthRecovery": {"title": "{hostname}: Resuelto - {category}{entity_suffix}", "body": "Una comprobación reciente confirma que la condición de {category} volvió a la normalidad.\nObservación anterior: {reason}\nGravedad anterior: {original_severity}\nTiempo desde la primera observación: {duration}", "status": "Resuelto"}, "backup": { "confirmedTitle": "{hostname}: backup completado", "confirmedBody": "Backup completado correctamente.", "errorTitle": "{hostname}: error notificado en el backup", "errorBody": "El informe del backup contiene un error.", - "unconfirmedBody": "El resultado del backup no está confirmado." + "unconfirmedBody": "El resultado del backup no está confirmado.", + "warningTitle": "{hostname}: Backup completado con advertencias", + "warningBody": "Backup completado con advertencias.", + "diagnosticsOmitted": "Líneas o texto de diagnóstico omitidos: {count}. Se conserva el informe original." } } } diff --git a/AppImage/messages/fr/common.json b/AppImage/messages/fr/common.json index d3ebeea9..a0ffa399 100644 --- a/AppImage/messages/fr/common.json +++ b/AppImage/messages/fr/common.json @@ -6980,7 +6980,8 @@ "failed": "Échec", "completed": "Terminé", "started": "Commencé", - "unconfirmed": "Non confirmé" + "unconfirmed": "Non confirmé", + "completed_with_warnings": "Terminée avec avertissements" }, "report": "Rapport {group}", "details": "Détails", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "Les relevés élevés s'étendent sur {duration}." }, + "healthRecovery": {"title": "{hostname} : Résolu - {category}{entity_suffix}", "body": "Un contrôle récent confirme le retour à la normale de la condition {category}.\nObservation précédente : {reason}\nGravité précédente : {original_severity}\nTemps depuis la première observation : {duration}", "status": "Résolu"}, "backup": { "confirmedTitle": "{hostname} : sauvegarde terminée", "confirmedBody": "Sauvegarde terminée avec succès.", "errorTitle": "{hostname} : erreur signalée lors de la sauvegarde", "errorBody": "Le rapport de sauvegarde contient une erreur.", - "unconfirmedBody": "Le résultat de la sauvegarde n’est pas confirmé." + "unconfirmedBody": "Le résultat de la sauvegarde n’est pas confirmé.", + "warningTitle": "{hostname}: Sauvegarde terminée avec avertissements", + "warningBody": "Sauvegarde terminée avec avertissements.", + "diagnosticsOmitted": "Lignes ou texte de diagnostic omis : {count}. Rapport original conservé." } } } diff --git a/AppImage/messages/it/common.json b/AppImage/messages/it/common.json index 2b7f0990..456091d1 100644 --- a/AppImage/messages/it/common.json +++ b/AppImage/messages/it/common.json @@ -6980,7 +6980,8 @@ "failed": "Fallito", "completed": "Completato", "started": "Iniziato", - "unconfirmed": "Non confermato" + "unconfirmed": "Non confermato", + "completed_with_warnings": "Completato con avvisi" }, "report": "{group} Rapporto", "details": "Dettagli", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "Intervallo dei campioni sopra soglia: {duration}." }, + "healthRecovery": {"title": "{hostname}: Risolto - {category}{entity_suffix}", "body": "Un controllo recente conferma che la condizione {category} è tornata nella norma.\nOsservazione precedente: {reason}\nGravità precedente: {original_severity}\nTempo dalla prima osservazione: {duration}", "status": "Risolto"}, "backup": { "confirmedTitle": "{hostname}: backup completato", "confirmedBody": "Backup completato correttamente.", "errorTitle": "{hostname}: errore segnalato nel backup", "errorBody": "Il rapporto del backup contiene un errore.", - "unconfirmedBody": "L’esito del backup non è confermato." + "unconfirmedBody": "L’esito del backup non è confermato.", + "warningTitle": "{hostname}: Backup completato con avvisi", + "warningBody": "Backup completato con avvisi.", + "diagnosticsOmitted": "Righe o testo diagnostico omessi: {count}. Il report originale è conservato." } } } diff --git a/AppImage/messages/pt/common.json b/AppImage/messages/pt/common.json index 98580f03..6b731ed2 100644 --- a/AppImage/messages/pt/common.json +++ b/AppImage/messages/pt/common.json @@ -6980,7 +6980,8 @@ "failed": "Falhou", "completed": "Concluído", "started": "Iniciado", - "unconfirmed": "Não confirmado" + "unconfirmed": "Não confirmado", + "completed_with_warnings": "Concluído com avisos" }, "report": "Relatório {group}", "details": "Detalhes", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "As amostras elevadas abrangem {duration}." }, + "healthRecovery": {"title": "{hostname}: Resolvido - {category}{entity_suffix}", "body": "Uma verificação recente confirma que a condição de {category} voltou ao normal.\nObservação anterior: {reason}\nGravidade anterior: {original_severity}\nTempo desde a primeira observação: {duration}", "status": "Resolvido"}, "backup": { "confirmedTitle": "{hostname}: backup concluído", "confirmedBody": "Backup concluído com sucesso.", "errorTitle": "{hostname}: erro comunicado no backup", "errorBody": "O relatório do backup contém um erro.", - "unconfirmedBody": "O resultado do backup não está confirmado." + "unconfirmedBody": "O resultado do backup não está confirmado.", + "warningTitle": "{hostname}: Backup concluído com avisos", + "warningBody": "Backup concluído com avisos.", + "diagnosticsOmitted": "Linhas ou texto de diagnóstico omitidos: {count}. Relatório original preservado." } } } diff --git a/AppImage/messages/sv/common.json b/AppImage/messages/sv/common.json index 405e1553..d1842183 100644 --- a/AppImage/messages/sv/common.json +++ b/AppImage/messages/sv/common.json @@ -6980,7 +6980,8 @@ "failed": "Misslyckades", "completed": "Klar", "started": "Startat", - "unconfirmed": "Obekräftat" + "unconfirmed": "Obekräftat", + "completed_with_warnings": "Klar med varningar" }, "report": "{group} Rapportera", "details": "Detaljer", @@ -6991,12 +6992,16 @@ "temperature": { "sampleSpan": "De höga mätvärdena sträcker sig över {duration}." }, + "healthRecovery": {"title": "{hostname}: Åtgärdat - {category}{entity_suffix}", "body": "En aktuell hälsokontroll bekräftar att tillståndet för {category} återgått till det normala.\nTidigare observation: {reason}\nTidigare allvarlighetsgrad: {original_severity}\nTid sedan första observationen: {duration}", "status": "Åtgärdat"}, "backup": { "confirmedTitle": "{hostname}: säkerhetskopiering klar", "confirmedBody": "Säkerhetskopieringen slutfördes utan fel.", "errorTitle": "{hostname}: fel rapporterat vid säkerhetskopiering", "errorBody": "Rapporten om säkerhetskopieringen innehåller ett fel.", - "unconfirmedBody": "Säkerhetskopieringens resultat är inte bekräftat." + "unconfirmedBody": "Säkerhetskopieringens resultat är inte bekräftat.", + "warningTitle": "{hostname}: Säkerhetskopiering klar med varningar", + "warningBody": "Säkerhetskopiering klar med varningar.", + "diagnosticsOmitted": "Utelämnade diagnosrader eller text: {count}. Originalrapporten bevaras." } } } diff --git a/AppImage/scripts/health_monitor.py b/AppImage/scripts/health_monitor.py index 26d38470..565f68e3 100644 --- a/AppImage/scripts/health_monitor.py +++ b/AppImage/scripts/health_monitor.py @@ -1452,7 +1452,11 @@ class HealthMonitor: status = 'OK' reason = None # CPU is normal - auto-resolve any existing CPU errors - health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal') + evidence = None + if cpu_percent < self.CPU_RECOVERY and len(recovery_samples) >= RECOVERY_MIN_SAMPLES: + evidence = {'check': 'cpu_usage', 'checked_at': current_time} + health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal', + check_evidence=evidence) temp_status = self._check_cpu_temperature() diff --git a/AppImage/scripts/health_persistence.py b/AppImage/scripts/health_persistence.py index 4fb13fc7..35458418 100644 --- a/AppImage/scripts/health_persistence.py +++ b/AppImage/scripts/health_persistence.py @@ -744,12 +744,12 @@ class HealthPersistence: return event_info - def resolve_error(self, error_key: str, reason: str = 'auto-resolved'): + def resolve_error(self, error_key: str, reason: str = 'auto-resolved', *, check_evidence=None): """Mark an error as resolved""" with self._db_lock: - return self._resolve_error_impl(error_key, reason) + return self._resolve_error_impl(error_key, reason, check_evidence=check_evidence) - def _resolve_error_impl(self, error_key, reason): + def _resolve_error_impl(self, error_key, reason, *, check_evidence=None): with self._db_connection() as conn: cursor = conn.cursor() now = datetime.now().isoformat() @@ -788,12 +788,59 @@ class HealthPersistence: stored_details = None self._record_event(cursor, 'resolved', error_key, { 'reason': reason, + # Only explicit current-check callers attach this proof. + # Generic resolve/cleanup remains neutral. + 'check_evidence': check_evidence, 'entity': self._entity_from_details(stored_details), 'details': stored_details or {}, }) conn.commit() + def get_recovery_evidence(self, error_key: str, first_seen: str): + """Return fresh same-incident native check proof, never absence of errors. + + Initially only the host CPU check has a stable condition identity. Other + checks, generic clears, excluded/deleted records and legacy events stay + neutral until they have equivalent per-condition provenance. + """ + if error_key != 'cpu_usage' or not first_seen: + return None + try: + with self._db_connection() as conn: + row = conn.execute(''' + SELECT first_seen, last_seen, resolved_at, acknowledged + FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1 + ''', (error_key,)).fetchone() + if not row or row[0] != first_seen or not row[2] or row[3]: + return None + event = conn.execute(''' + SELECT timestamp, data FROM events + WHERE error_key = ? AND event_type = 'resolved' + ORDER BY id DESC LIMIT 1 + ''', (error_key,)).fetchone() + if not event: + return None + proof = json.loads(event[1]).get('check_evidence') + if not isinstance(proof, dict) or proof.get('check') != error_key: + return None + checked = proof.get('checked_at') + if not isinstance(checked, (int, float)) or isinstance(checked, bool): + return None + checked = float(checked) + # Reuse the collector's existing two-hour freshness boundary. + now = datetime.now().timestamp() + if not 0 <= now - checked <= 7200: + return None + last_seen = datetime.fromisoformat(row[1]).timestamp() + resolved = datetime.fromisoformat(row[2]).timestamp() + recorded = datetime.fromisoformat(event[0]).timestamp() + if not last_seen <= checked <= resolved <= recorded: + return None + return proof + except (ValueError, TypeError, AttributeError, OverflowError): + return None + def is_error_active(self, error_key: str, category: Optional[str] = None) -> bool: """ Check if an error is currently active OR suppressed (dismissed but within suppression period). diff --git a/AppImage/scripts/notification_channels.py b/AppImage/scripts/notification_channels.py index bf74ddeb..b0b106ab 100644 --- a/AppImage/scripts/notification_channels.py +++ b/AppImage/scripts/notification_channels.py @@ -1038,13 +1038,23 @@ class EmailChannel(NotificationChannel): # Determine group for section header event_type = data.get('_event_type', '') if event_type == 'error_resolved': - sev.update(self._SEV_DEFAULT) - sev['label'] = _runtime_text('email.severity.observation', data) + if (data.get('recovery_outcome') == 'resolved' + and data.get('is_recovery') is True + and isinstance(data.get('check_evidence'), dict) + and data['check_evidence'].get('check') == 'cpu_usage'): + sev.update(self._SEV_STYLE['OK']) + sev['label'] = _runtime_notification_text('healthRecovery.status', data) + else: + sev.update(self._SEV_DEFAULT) + sev['label'] = _runtime_text('email.severity.observation', data) elif event_type == 'backup_complete': outcome = data.get('backup_outcome') if outcome == 'confirmed': sev.update(self._SEV_STYLE['OK']) status = 'completed' + elif outcome == 'completed_with_warnings': + sev.update(self._SEV_STYLE['WARNING']) + status = 'completed_with_warnings' elif outcome == 'failed': sev.update(self._SEV_STYLE['CRITICAL']) status = 'failed' @@ -1057,7 +1067,7 @@ class EmailChannel(NotificationChannel): # bodies, including restore bodies released from quiet hours. backup_email = event_type in {'backup_complete', 'backup_fail'} wrap_body = (event_type in {'temp_high', 'system_restore_completed', 'error_resolved'} - or backup_email or data.get('_restore_summary')) + or backup_email or data.get('_restore_summary') or data.get('_backup_summary')) temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else '' temp_table_layout = 'table-layout:fixed;' if wrap_body else '' backup_title_wrap = temp_cell_wrap if backup_email else '' @@ -1103,7 +1113,7 @@ class EmailChannel(NotificationChannel): # Observation age/disappearance must not become a green OK row. # The endpoint's warnings_block and task counts live in the # localized body, not the generic services Event row. - detail_rows = [('', html_mod.escape(line.strip())) + detail_rows = [('', html_mod.escape(line if data.get('_quiet_hours_summary') else line.strip())) for line in body.split('\n') if line.strip()] # ── Fallback: if no structured rows, render body text lines ── @@ -1122,6 +1132,7 @@ class EmailChannel(NotificationChannel): # ── Render detail rows as HTML table ── rows_html = '' + summary_whitespace = 'white-space:pre-wrap;' if data.get('_quiet_hours_summary') and data.get('_restore_summary') else '' for label, value in detail_rows: if label: rows_html += f''' @@ -1131,7 +1142,7 @@ class EmailChannel(NotificationChannel): else: # Full-width row (no label, just description text) rows_html += f''' - {value} + {value} ''' # ── Reason / details block (long text, displayed separately) ── @@ -1291,6 +1302,7 @@ class EmailChannel(NotificationChannel): _add('Storage', data.get('storage') or data.get('storage_name'), 'code') if event_type == 'backup_complete' and data.get('backup_outcome') != 'confirmed': status_key = ('failed' if data.get('backup_outcome') == 'failed' + else 'completed_with_warnings' if data.get('backup_outcome') == 'completed_with_warnings' else 'unconfirmed') else: status_key = 'failed' if 'fail' in event_type else 'completed' if 'complete' in event_type else 'started' diff --git a/AppImage/scripts/notification_events.py b/AppImage/scripts/notification_events.py index 4aa91067..9aedaf3a 100644 --- a/AppImage/scripts/notification_events.py +++ b/AppImage/scripts/notification_events.py @@ -3223,6 +3223,13 @@ class PollingCollector: self._last_notified.pop(key, None) continue + # Disappearance is not recovery. Only same-incident proof written + # by a successful existing native check can certify normality. + try: + recovery_evidence = health_persistence.get_recovery_evidence(key, first_seen) + except Exception: + recovery_evidence = None + # Calculate duration duration = '' if first_seen: @@ -3273,6 +3280,9 @@ class PollingCollector: else: clean_reason = 'Condition no longer reported' + if recovery_evidence: + clean_reason = reason_summary + # `original_severity` must match what the user actually saw # in the most-recent notification for this error, not the # latest DB severity. See `_notified_severity` docstring at @@ -3299,7 +3309,9 @@ class PollingCollector: 'original_severity': original_severity, 'first_seen': first_seen, 'duration': duration_label, - 'is_recovery': True, + 'is_recovery': bool(recovery_evidence), + 'recovery_outcome': 'resolved' if recovery_evidence else 'no_longer_reported', + 'check_evidence': recovery_evidence, } # Spread the original details blob so the resolved notification # can use the same {storage_name}/{vm_name}/{device} placeholders @@ -4302,14 +4314,16 @@ class ProxmoxHookWatcher: table = _parse_vzdump_table(text) if table is not None and any(guest['status'].lower() == 'error' for guest in table['vms']): return 'failed' - if severity not in ('info', 'ok', 'success') or re.search( - r'(?im)(?:^\s*(?:\d+:\s*)?(?:\d{4}-\d{2}-\d{2}\s+\S+\s+)?WARN(?:ING)?:|\bWARNINGS\s*:\s*\d+)', text): + warnings = severity in ('warning', 'warn') or bool(re.search( + r'(?im)(?:^\s*(?:\d+:\s*)?(?:\d{4}-\d{2}-\d{2}\s+\S+\s+)?WARN(?:ING)?:|\bWARNINGS\s*:\s*[1-9]\d*)', text)) + if severity not in ('info', 'ok', 'success', 'warning', 'warn'): return 'unconfirmed' + completed = 'completed_with_warnings' if warnings else 'confirmed' # A present table is authoritative: do not certify an incomplete table # from a finished guest log, or reject a complete OK table merely # because the extra diagnostic log was truncated before its finishes. if table is not None: - return ('confirmed' if table['complete'] and + return (completed if table['complete'] and all(guest['status'].lower() == 'ok' for guest in table['vms']) else 'unconfirmed') if starts: @@ -4322,10 +4336,10 @@ class ProxmoxHookWatcher: return 'unconfirmed' # A finish before its start is not evidence. else: pending[vmid] -= 1 - return 'confirmed' if not any(pending.values()) else 'unconfirmed' + return completed if not any(pending.values()) else 'unconfirmed' if re.search( r'(?im)^\s*(?:INFO:\s*)?TASK OK\s*$', text): - return 'confirmed' + return completed return 'unconfirmed' def process_webhook(self, payload: dict) -> dict: diff --git a/AppImage/scripts/notification_manager.py b/AppImage/scripts/notification_manager.py index 7f317314..56abc814 100644 --- a/AppImage/scripts/notification_manager.py +++ b/AppImage/scripts/notification_manager.py @@ -1364,6 +1364,12 @@ class NotificationManager: # Get journal context if available (will be enriched per-channel based on detail_level) raw_journal_context = data.get('_journal_context', '') + # Persist a presentation token in the existing title column: buffers + # otherwise discard outcome metadata before composition. Old rows + # without a token remain neutral; routing and schema are unchanged. + buffer_title = title + if event_type in ('backup_complete', 'backup_fail'): + buffer_title, _ = enrich_with_emojis(event_type, title, '', data) for ch_name, channel in channels.items(): # ── Per-channel category check ── @@ -1393,7 +1399,7 @@ class NotificationManager: # delivered after Quiet Hours + Daily Digest were merged. if severity != 'CRITICAL' and self._in_quiet_hours(ch_name): self._buffer_quiet_event(ch_name, event_type, event_group, - severity, title, body) + severity, buffer_title, body) continue # ── Per-channel daily digest ── @@ -1406,7 +1412,7 @@ class NotificationManager: # excluded from the digest by `_DIGEST_EXEMPT_EVENTS`. if self._should_buffer_for_digest(ch_name, severity, event_type): self._buffer_digest_event(ch_name, event_type, event_group, - severity, title, body) + severity, buffer_title, body) continue try: @@ -1849,14 +1855,22 @@ class NotificationManager: if index < 8 or (quiet_release and item[1] == 'system_restore_completed')] for ts, ev_type, title, body in visible_items: hhmm = datetime.fromtimestamp(ts).strftime('%H:%M') + backup_icon = '' + if ev_type in ('backup_complete', 'backup_fail'): + for token in ('💾✅', '💾⚠️', '💾❌', '💾❔', '💾'): + if title.startswith(token + ' '): + backup_icon = token + title = title[len(token) + 1:] + break + backup_icon = backup_icon or ('💾❌' if ev_type == 'backup_fail' else '💾❔') short_title = title.split(': ', 1)[-1] if ': ' in title else title event_icon = ( - EVENT_EMOJI.get(ev_type) or CATEGORY_EMOJI.get(group, '') + backup_icon or EVENT_EMOJI.get(ev_type) or CATEGORY_EMOJI.get(group, '') ) if use_icons else '' event_prefix = f'{event_icon} ' if event_icon else '' lines.append(f" • {event_prefix}{hhmm} {short_title}") if quiet_release and ev_type == 'system_restore_completed' and body: - lines.extend(line for line in body.splitlines() if line.strip()) + lines.extend(' ' + line.strip() for line in body.splitlines() if line.strip()) if len(items) > len(visible_items): lines.append(runtime_message('digest.more', language, count=len(items) - len(visible_items))) lines.append('') @@ -2006,6 +2020,7 @@ class NotificationManager: summary_title, summary_body, severity='INFO', data={'_quiet_hours_summary': True, '_count': len(rows), '_restore_summary': any(row[1] == 'system_restore_completed' for row in rows), + '_backup_summary': any(row[1] in ('backup_complete', 'backup_fail') for row in rows), '_notification_language': language}, ) or result except Exception as e: diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index 47dbc221..6bb5f513 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -244,6 +244,13 @@ def _parse_vzdump_table(message: str) -> Optional[Dict[str, Any]]: status = 'error' kind = ('lxc' if 'lxc' in filename or filename.startswith('ct/') else 'qemu' if 'qemu' in filename or filename.startswith('vm/') else '') + if filename.lower() == 'null': + # Native failed rows lack archives. Match only this row's VMID; + # multiple inconsistent starts are not authoritative identity. + kinds = set(re.findall( + r'(?im)\bStarting Backup of VM ' + re.escape(vmid) + r'\s+\((lxc|qemu)\)', + message)) + kind = kinds.pop() if len(kinds) == 1 else '' rows.append({'vmid': vmid, 'name': name, 'status': status, 'time': duration, 'size': size, 'filename': filename, 'type': kind}) @@ -1892,6 +1899,10 @@ def render_template(event_type: str, data: Dict[str, Any], template['title'] = runtime_message('backup.confirmedTitle', language, hostname=data.get('hostname') or _get_hostname()) template['body'] = runtime_message('backup.confirmedBody', language) + elif outcome == 'completed_with_warnings': + template['title'] = runtime_message('backup.warningTitle', language, + hostname=data.get('hostname') or _get_hostname()) + template['body'] = runtime_message('backup.warningBody', language) elif outcome == 'failed': template['title'] = runtime_message('backup.errorTitle', language, hostname=data.get('hostname') or _get_hostname()) @@ -1899,7 +1910,7 @@ def render_template(event_type: str, data: Dict[str, Any], if event_type == 'backup_fail': template['title'] = runtime_message('backup.errorTitle', language, hostname=data.get('hostname') or _get_hostname()) - if event_type == 'backup_fail' or (event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'failed')): + if event_type == 'backup_fail' or (event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'completed_with_warnings', 'failed')): parsed_backup = _parse_vzdump_message(str(data.get('pve_message') or '')) storage = str((parsed_backup or {}).get('storage_name') or data.get('storage') or '').strip() guests = (parsed_backup or {}).get('vms') or [] @@ -1950,7 +1961,7 @@ def render_template(event_type: str, data: Dict[str, Any], 'log_file': '', } variables.update(data) - if event_type == 'backup_fail' or (event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'failed')): + if event_type == 'backup_fail' or (event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'completed_with_warnings', 'failed')): # The provider has already substituted raw Display Names. Insert the # resolved title as a value, never reinterpret its literal braces. variables['_backup_title'] = template['title'] @@ -2059,6 +2070,14 @@ def render_template(event_type: str, data: Dict[str, Any], return '' safe_vars = _SafeDict(variables) + if (event_type == 'error_resolved' and data.get('recovery_outcome') == 'resolved' + and data.get('is_recovery') is True + and isinstance(data.get('check_evidence'), dict) + and data['check_evidence'].get('check') == 'cpu_usage'): + safe_vars['_health_title'] = runtime_message('healthRecovery.title', language, **variables) + safe_vars['_health_body'] = runtime_message('healthRecovery.body', language, **variables) + template['title'] = '{_health_title}' + template['body'] = '{_health_body}' try: title = template['title'].format_map(safe_vars) except (ValueError, IndexError): @@ -2069,6 +2088,30 @@ def render_template(event_type: str, data: Dict[str, Any], # When the event came from PVE webhook with a full vzdump message, # parse the table/logs and format a rich body instead of the sparse template. pve_message = data.get('pve_message', '') + backup_diagnostics = [] + + def bounded_backup_diagnostics(lines): + # 1024 chars matches the repository's small-channel message convention; + # 8 lines keeps repeated producer warnings readable. Inventory/title + # size is separate: this is not a one-Telegram-message guarantee. + unique = list(dict.fromkeys(line for line in lines if line.strip())) + principal = next((line for line in unique if re.search(r'\b(?:ERROR:|TASK ERROR:)', line, re.IGNORECASE)), None) + if principal: + unique.remove(principal) + unique.insert(0, principal) + shown, budget, omitted = [], 1024, 0 + for line in unique: + if len(shown) >= 8 or budget < 2: + omitted += 1 + continue + line_budget = min(budget, 512) + rendered = line if len(line) <= line_budget else line[:line_budget - 1] + '…' + omitted += int(rendered != line) + shown.append(rendered) + budget -= len(rendered) + 1 + if omitted: + shown.append(runtime_message('backup.diagnosticsOmitted', language, count=omitted)) + return '\n'.join(shown) # Check for custom formatter function formatter_name = template.get('formatter') @@ -2099,15 +2142,15 @@ def render_template(event_type: str, data: Dict[str, Any], pve_message, re.MULTILINE | re.DOTALL) if error_block: diagnostic_lines = error_block.group(1).rstrip('\r\n').splitlines() + diagnostic_lines - if diagnostic_lines: - body_text += '\n' + '\n'.join( - line for line in dict.fromkeys(diagnostic_lines) - if line.strip() and line not in body_text.splitlines()) + backup_diagnostics = [line for line in diagnostic_lines + if line.strip() and line not in body_text.splitlines()] else: - # Couldn't parse -- use PVE raw message as body - body_text = pve_message.strip() + # Unparsed diagnostic-only reports remain visible but bounded. + body_text = '' + backup_diagnostics = pve_message.strip().splitlines() if event_type == 'backup_complete' and data.get('backup_outcome') != 'confirmed': key = ('backup.errorBody' if data.get('backup_outcome') == 'failed' + else 'backup.warningBody' if data.get('backup_outcome') == 'completed_with_warnings' else 'backup.unconfirmedBody') body_text = runtime_message(key, language) + '\n' + body_text elif event_type == 'system_mail' and pve_message: @@ -2138,8 +2181,20 @@ def render_template(event_type: str, data: Dict[str, Any], if event_type in ('backup_complete', 'backup_fail') and ( event_type == 'backup_fail' or data.get('backup_outcome') == 'failed'): source_subject = str(data.get('pve_title') or '').strip() + guest_context = (_parse_vzdump_message(str(pve_message or '')) or {}).get('vms') + if guest_context: + # Native single-line job errors live only in the subject. Retain + # that cause, not the redundant job/host envelope or generic count. + cause = re.search(r'\bbackup failed:\s*(.+)', source_subject, re.IGNORECASE) + source_subject = cause.group(1).strip() if cause else '' + if source_subject.lower() == 'multiple problems': + source_subject = '' if source_subject and source_subject not in body_text: - body_text += '\n' + source_subject + # A unique job/setup cause must survive a warning-heavy report. + backup_diagnostics.insert(0, source_subject) + + if backup_diagnostics: + body_text += '\n' + bounded_backup_diagnostics(backup_diagnostics) # Clean up: collapse runs of 3+ blank lines into 1, remove trailing whitespace import re as _re @@ -2472,9 +2527,14 @@ def enrich_with_emojis(event_type: str, title: str, body: str, severity = data.get('severity', 'INFO') icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '') + if (event_type == 'error_resolved' and data.get('recovery_outcome') == 'resolved' + and data.get('is_recovery') is True + and isinstance(data.get('check_evidence'), dict) + and data['check_evidence'].get('check') == 'cpu_usage'): + icon = '✅' if event_type == 'backup_complete': icon = { - 'confirmed': '💾✅', 'failed': '💾❌', + 'confirmed': '💾✅', 'completed_with_warnings': '💾⚠️', 'failed': '💾❌', }.get(str(data.get('backup_outcome') or ''), '💾❔') # Build enriched title: replace severity circle with event-specific icon diff --git a/AppImage/scripts/tests/test_notification_runtime_i18n.py b/AppImage/scripts/tests/test_notification_runtime_i18n.py index 6ac97b01..cf98806c 100644 --- a/AppImage/scripts/tests/test_notification_runtime_i18n.py +++ b/AppImage/scripts/tests/test_notification_runtime_i18n.py @@ -74,7 +74,10 @@ class RuntimeCatalogTests(unittest.TestCase): en = flatten(self.catalogs["en"]) pending_slovak = {"backup.confirmedTitle", "backup.confirmedBody", "backup.errorTitle", "backup.errorBody", "backup.unconfirmedBody", - "channels.email.severity.observation", "channels.email.status.unconfirmed"} + "channels.email.severity.observation", "channels.email.status.unconfirmed", + "backup.warningTitle", "backup.warningBody", "backup.diagnosticsOmitted", + "channels.email.status.completed_with_warnings", + "healthRecovery.title", "healthRecovery.body", "healthRecovery.status"} for language, catalog in self.catalogs.items(): translated = flatten(catalog) if language == 'sk': From 8e396f957fb13e847cc9909be1fd46a7ee3109cc Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Wed, 30 Sep 2026 23:04:53 +0200 Subject: [PATCH 09/14] fix(notifications): avoid repeating subject causes already in error diagnostics --- .../tests/test_notification_maintainer_followup.py | 8 ++++++++ AppImage/scripts/notification_templates.py | 6 +++++- 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/.github/scripts/tests/test_notification_maintainer_followup.py b/.github/scripts/tests/test_notification_maintainer_followup.py index 42d37377..74bfc2b9 100644 --- a/.github/scripts/tests/test_notification_maintainer_followup.py +++ b/.github/scripts/tests/test_notification_maintainer_followup.py @@ -129,6 +129,14 @@ class MaintainerFollowupTests(unittest.TestCase): self.assertIn('white-space:pre-wrap;',result['html']) self.assertNotIn(templates.runtime_message('digest.footer',lang),result['body']) + def test_concrete_subject_cause_already_in_error_log_is_not_repeated(self): + event=receive(NATIVE_REPORT+'\n100: ERROR: job-end hook denied','error', + 'vzdump backup status (raw-host): backup failed: job-end hook denied') + for lang in LANGUAGES: + result=deliver(event.event_type,event.data,event.severity,lang) + self.assertEqual(result['text'].count('job-end hook denied'),1) + self.assertIn('100: ERROR: job-end hook denied',result['body']) + def test_job_level_subject_cause_survives_warning_cap(self): raw=NATIVE_REPORT+'\n'+'\n'.join('WARN: repeated diagnostic '+str(i) for i in range(80)) event=receive(raw,'error','vzdump backup status (raw-host): backup failed: job-end hook denied') diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index 6bb5f513..99ae05ff 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -2189,7 +2189,11 @@ def render_template(event_type: str, data: Dict[str, Any], source_subject = cause.group(1).strip() if cause else '' if source_subject.lower() == 'multiple problems': source_subject = '' - if source_subject and source_subject not in body_text: + cause_in_diagnostics = any( + line.strip() == source_subject or + re.split(r'\b(?:TASK ERROR:|ERROR:)\s*', line, maxsplit=1, flags=re.IGNORECASE)[-1].strip() == source_subject + for line in backup_diagnostics) + if source_subject and source_subject not in body_text and not cause_in_diagnostics: # A unique job/setup cause must survive a warning-heavy report. backup_diagnostics.insert(0, source_subject) From 3be15211b1180ccb7d566a68d4adc2a87c0e26a4 Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Thu, 1 Oct 2026 00:23:36 +0200 Subject: [PATCH 10/14] Fix principal cause caps and bind measured recovery to incident policy --- .../tests/notification_recovery_fixture.py | 86 +++++++ .../test_notification_recovery_corrections.py | 225 ++++++++++++++++++ .../test_notification_recovery_evidence.py | 138 +++++------ .../test_notification_review_corrections.py | 24 ++ AppImage/scripts/health_monitor.py | 28 ++- AppImage/scripts/health_persistence.py | 77 +++--- AppImage/scripts/health_recovery.py | 51 ++++ AppImage/scripts/notification_channels.py | 13 +- AppImage/scripts/notification_templates.py | 35 ++- 9 files changed, 537 insertions(+), 140 deletions(-) create mode 100644 .github/scripts/tests/notification_recovery_fixture.py create mode 100644 .github/scripts/tests/test_notification_recovery_corrections.py create mode 100644 .github/scripts/tests/test_notification_review_corrections.py create mode 100644 AppImage/scripts/health_recovery.py diff --git a/.github/scripts/tests/notification_recovery_fixture.py b/.github/scripts/tests/notification_recovery_fixture.py new file mode 100644 index 00000000..8e4f91d6 --- /dev/null +++ b/.github/scripts/tests/notification_recovery_fixture.py @@ -0,0 +1,86 @@ +"""Pinned, inert recovery review. No operational module imports or threads. +Run with python3 -S under bwrap --unshare-net; DBs and export use TMPDIR. +""" +import ast, contextlib, datetime, hashlib, io, json, os, pathlib, sqlite3, subprocess, sys, tarfile, tempfile, threading, types, typing +from unittest.mock import patch +REPO=pathlib.Path('/home/martino/projects/proxmox/notification-maintainer-followup') +BASE=datetime.datetime(2026,9,30,20,0,0).timestamp() +class Clock(datetime.datetime): + epoch=BASE + tick=0.0 + @classmethod + def now(cls,tz=None): + value=cls.fromtimestamp(cls.epoch,tz) + cls.epoch+=cls.tick + return value +TIME=types.SimpleNamespace(time=lambda:Clock.epoch) + +def extract(path,name,owner,ns): + tree=ast.parse(path.read_text()) + nodes=tree.body if owner is None else next(n.body for n in tree.body if isinstance(n,ast.ClassDef) and n.name==owner) + node=next(n for n in nodes if isinstance(n,(ast.FunctionDef,ast.AsyncFunctionDef)) and n.name==name) + node.decorator_list=[] + exec(compile(ast.Module(body=[node],type_ignores=[]),str(path),'exec'),ns) + return ns[name] + +from notification_fixture import SCRIPTS as scripts +pns=dict(vars(typing),datetime=Clock,timedelta=datetime.timedelta,json=json,sqlite3=sqlite3,contextmanager=contextlib.contextmanager,re=__import__('re'),_re_disk_base=__import__('re')) +pns['disk_base_name']=extract(scripts/'health_persistence.py','disk_base_name',None,pns) +methods={n:extract(scripts/'health_persistence.py',n,'HealthPersistence',pns) for n in ('_get_conn','_db_connection','_init_database','record_error','_record_error_impl','resolve_error','_resolve_error_impl','get_recovery_evidence','_record_event','_entity_from_details','clear_error','get_active_errors','is_error_active','is_error_acknowledged','_get_setting_impl','get_setting','set_setting','acknowledge_error','_acknowledge_error_impl','get_excluded_interface_names')} +def make_store(directory): + s=types.SimpleNamespace(db_path=pathlib.Path(directory)/'health.sqlite',_db_lock=threading.RLock(),DEFAULT_SUPPRESSION_HOURS=24,CATEGORY_SETTING_MAP={}) + for n,f in methods.items(): + if n=='_entity_from_details':setattr(s,n,f) + elif n=='_db_connection':setattr(s,n,types.MethodType(contextlib.contextmanager(f),s)) + else:setattr(s,n,types.MethodType(f,s)) + s._init_database() + return s +def sql(s,query,args=()): + with s._db_connection() as c: + data=c.execute(query,args).fetchall();c.commit();return data +def record(s,key='cpu_usage',category='cpu',reason='CPU high',details=None): + Clock.epoch=BASE-600 + with patch.dict(sys.modules,{'os':types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:True))}):s.record_error(key,category,'WARNING',reason,details) + Clock.epoch=BASE-10 + with patch.dict(sys.modules,{'os':types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:True))}):s.record_error(key,category,'WARNING',reason,details) + Clock.epoch=BASE + assert s.get_active_errors() + return sql(s,'SELECT first_seen FROM errors WHERE error_key=?',(key,))[0][0] +def cpu(s,current=20,history=None,warning=85,critical=95): + ns=dict(vars(typing),time=TIME,os=types.SimpleNamespace(cpu_count=lambda:4),health_persistence=s,psutil=types.SimpleNamespace(cpu_percent=lambda **kw:current,cpu_count=lambda:4)) + fn=extract(scripts/'health_monitor.py','_check_cpu_with_hysteresis','HealthMonitor',ns) + if history is None:history=[{'value':20,'time':BASE-i*10} for i in range(1,11)] + target=types.SimpleNamespace(state_history={'cpu_usage':list(history)},CPU_WARNING=85,CPU_CRITICAL=95,CPU_RECOVERY=75,CPU_WARNING_DURATION=300,CPU_CRITICAL_DURATION=300,CPU_RECOVERY_DURATION=120,_check_cpu_temperature=lambda:None) + refresh=extract(scripts/'health_monitor.py','_refresh_thresholds','HealthMonitor',{}) + with patch.dict(sys.modules,{'health_thresholds':types.SimpleNamespace(get=lambda section,key:({'warning':warning,'critical':critical}.get(key) if section=='cpu' else None))}):refresh(target) + return fn(target) +def poll(s,first,key='cpu_usage',category='cpu',reason='CPU high',details=None,first_done=True,foreign=False,restored=False): + events=[] + meta={'category':category,'reason':reason,'severity':'WARNING','first_seen':first,'details':details} + c=types.SimpleNamespace(_hostname='node-a',_ENTITY_MAP={'cpu':('node',''),'pve_services':('node',''),'network':('node','')},_first_poll_done=first_done,_known_errors={key:meta},_notified_severity={key:'WARNING'},_last_notified={key:BASE-1},SAME_ERROR_COOLDOWN=86400,_get_cooldown_from_db=lambda *a:BASE-1,_queue=types.SimpleNamespace(put=events.append),_guest_storage_error_is_now_foreign=lambda *a:foreign,_save_known_errors_meta=lambda:None) + ns=dict(vars(typing),time=TIME,json=json,re=__import__('re'),NotificationEvent=lambda *a,**kw:types.SimpleNamespace(event_type=a[0],severity=a[1],data=a[2],**kw),startup_grace=types.SimpleNamespace(should_suppress_category=lambda *a:False)) + c._guest_storage_error_is_now_foreign=extract(scripts/'notification_events.py','_guest_storage_error_is_now_foreign','PollingCollector',ns) + fn=extract(scripts/'notification_events.py','_check_persistent_health','PollingCollector',ns) + with patch.dict(sys.modules,{'health_persistence':types.SimpleNamespace(health_persistence=s),'flask_server':types.SimpleNamespace(get_proxmox_node_name=lambda:'node-a',get_cached_pvesh_cluster_resources_vm=lambda:[{'vmid':100,'type':'lxc','node':'node-b' if foreign else 'node-a'}]),'datetime':types.SimpleNamespace(**{**vars(datetime),'datetime':Clock})}): + if restored: + c._KNOWN_ERRORS_SETTING_KEY='pollingcollector_known_errors_v1' + s.set_setting(c._KNOWN_ERRORS_SETTING_KEY,json.dumps(c._known_errors));c._known_errors={} + extract(scripts/'notification_events.py','_load_known_errors_meta','PollingCollector',ns)(c) + c._first_poll_done=bool(c._known_errors) + fn(c) + return [e.data for e in events],c +def service(store, rc=0, stdout='active\n', raised=False, services=('pvedaemon',), clustered=False): + calls=[] + def run(argv, **kw): + calls.append((argv, kw)) + assert argv[:2] == ['systemctl', 'is-active'] + if raised: raise TimeoutError('inert timeout') + return types.SimpleNamespace(returncode=rc, stdout=stdout) + ns=dict(vars(typing),time=TIME,os=types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:clustered)),subprocess=types.SimpleNamespace(run=run),health_persistence=store) + result=extract(scripts/'health_monitor.py','_check_pve_services','HealthMonitor',ns)(types.SimpleNamespace(PVE_SERVICES=list(services))) + return result,calls + +@contextlib.contextmanager +def case(): + Clock.epoch=BASE + with tempfile.TemporaryDirectory(prefix='recovery-review-db-',dir=os.environ['TMPDIR']) as d:yield make_store(d) diff --git a/.github/scripts/tests/test_notification_recovery_corrections.py b/.github/scripts/tests/test_notification_recovery_corrections.py new file mode 100644 index 00000000..6908cbfb --- /dev/null +++ b/.github/scripts/tests/test_notification_recovery_corrections.py @@ -0,0 +1,225 @@ +"""Native initializer, measurement methods, SQL writers/readers and collector.""" +import unittest +from notification_recovery_fixture import case, Clock, BASE, cpu, sql, poll + +class RecoveryCorrectionTests(unittest.TestCase): + def test_supported_low_warning_current_violation_is_neutral(self): + with case() as store: + Clock.epoch = BASE - 600 + initial = cpu(store, 80, [{'value':80, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=50) + self.assertEqual(initial['status'], 'WARNING') + first = sql(store, 'SELECT first_seen FROM errors')[0][0] + Clock.epoch = BASE + result = cpu(store, 60, [{'value':60, 'time':BASE-i*5} for i in range(1,26)], warning=50) + events, _ = poll(store, first, reason=initial['reason']) + # Preserve operational clear/hysteresis behavior, not its factual claim. + self.assertEqual(result['status'], 'OK') + self.assertFalse(events[0]['is_recovery'], events) + self.assertIsNone(store.get_recovery_evidence('cpu_usage', first)) + + def test_original_policy_survives_repeated_native_updates(self): + import json + for later_warning in (85, 40): + with self.subTest(later_warning=later_warning), case() as store: + Clock.epoch = BASE - 600 + cpu(store, 80, [{'value':80, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=50) + first = sql(store, 'SELECT first_seen FROM errors')[0][0] + for step in (400, 200): + Clock.epoch = BASE-step + cpu(store, 90, [{'value':90, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=later_warning) + original = json.loads(sql(store, 'SELECT details FROM errors')[0][0])['cpu_policy'] + self.assertEqual(original['warning'], 50) + Clock.epoch = BASE + cpu(store, 20, warning=later_warning) + self.assertIsNone(store.get_recovery_evidence('cpu_usage', first)) + self.assertFalse(poll(store, first)[0][0]['is_recovery']) + + def test_clock_rollback_new_row_generic_clear_cannot_inherit_old_proof(self): + for rollback, reuse_first in ((True,False),(False,False),(True,True)): + with self.subTest(rollback=rollback,reuse_first=reuse_first),case() as store: + Clock.epoch = BASE-600 + cpu(store, 90, [{'value':90, 'time':Clock.epoch-i*5} for i in range(1,26)]) + first = sql(store, 'SELECT first_seen FROM errors')[0][0] + Clock.epoch = BASE; Clock.tick = .001 + try: cpu(store) + finally: Clock.tick = 0 + self.assertTrue(store.get_recovery_evidence('cpu_usage', first)) + store.acknowledge_error('cpu_usage', suppression_hours=-1) + store.clear_error('cpu_usage') + Clock.epoch = BASE-100 if rollback else BASE+1 + store.record_error('cpu_usage','cpu','WARNING','new incident after clock step',{}) + second = sql(store, 'SELECT first_seen FROM errors')[0][0] + self.assertNotEqual(first, second) + if reuse_first: + # Restored malformed snapshot with reused wall-clock identity; + # native row id and latest closure still prevent replay. + sql(store,'UPDATE errors SET first_seen=?',(first,));second=first + Clock.epoch = BASE+.0005 if rollback else BASE+2 + store.clear_error('cpu_usage'); Clock.epoch = BASE+3 + self.assertIsNone(store.get_recovery_evidence('cpu_usage', second)) + self.assertFalse(poll(store, second)[0][0]['is_recovery']) + + def test_malformed_native_and_manual_proof_is_neutral_at_all_consumers(self): + import copy, json, time + from unittest.mock import patch + from notification_fixture import LANGUAGES + from notification_final_fixture import deliver + with case() as store: + Clock.epoch = BASE-600 + cpu(store, 90, [{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)]) + first = sql(store,'SELECT first_seen FROM errors')[0][0] + Clock.epoch = BASE; cpu(store) + event_data = json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0]) + data = poll(store,first)[0][0] + for field, bad in [('checked_at','bad'),('checked_at',True),('checked_at',float('inf')), + ('checked_at',10**400),('value',60),('max_sample',float('nan')), + ('normal_samples',True),('normal_samples',9),('checked_at',BASE+1),('checked_at',BASE-7201), + ('policy',{'warning':False,'critical':95,'recovery':75}), + ('policy',{'warning':96,'critical':95,'recovery':75}), + ('policy',{'warning':85,'critical':95}), + ('policy',{'warning':85,'critical':95,'recovery':float('nan')}), + ('policy',{'warning':85,'critical':95,'recovery':75,'extra':0})]: + broken = copy.deepcopy(event_data) + broken['check_evidence'][field] = bad + if field == 'value': broken['check_evidence']['policy']['warning'] = 50 + sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps(broken),)) + with self.subTest(field=field,bad=str(bad)),patch('health_recovery.time.time',return_value=BASE): + self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) + for language in LANGUAGES: + for manual in (False,True): + result = deliver('error_resolved',{**data,'check_evidence':broken['check_evidence']},'OK',language,manual=manual) + self.assertNotIn('background:#f0fdf4;', result['html']) + # Caller content is trusted, not authenticated native proof; even + # well-shaped assertions require a valid time/type/numeric contract. + for proof in ({'check':'cpu_usage'}, {'check':'cpu_usage','checked_at':time.time()+1}): + self.assertNotIn('background:#f0fdf4;', deliver('error_resolved',{**data,'check_evidence':proof},'OK',manual=True)['html']) + + def test_exact_service_active_native_clear_reaches_recovery_consumers(self): + from unittest.mock import patch + from notification_recovery_fixture import service + from notification_fixture import LANGUAGES + from notification_final_fixture import deliver + with case() as store: + Clock.epoch = BASE-600 + service(store,3,'inactive\n') + first = sql(store,'SELECT first_seen FROM errors')[0][0] + Clock.epoch = BASE + result, calls = service(store) + self.assertEqual(result['status'],'OK') + self.assertEqual(calls,[(['systemctl','is-active','pvedaemon'], {'capture_output':True,'text':True,'timeout':2})]) + proof = store.get_recovery_evidence('pve_service_pvedaemon',first) + self.assertTrue(proof) + data = poll(store,first,'pve_service_pvedaemon','pve_services','PVE service pvedaemon is inactive',details={'service':'pvedaemon'})[0][0] + self.assertTrue(data['is_recovery']) + with patch('health_recovery.time.time',return_value=BASE): + for language in LANGUAGES: + for manual in (False,True): + rendered = deliver('error_resolved',data,'OK',language,manual=manual) + self.assertIn('background:#f0fdf4;',rendered['html']) + self.assertIn('pvedaemon', rendered['text']) + + def test_cpu_positive_default_low_policy_and_neutral_history_controls(self): + from notification_recovery_fixture import record + for warning,current,history,legacy,expected in ( + (85,20,None,False,True), (50,20,None,False,True), + (85,20,None,True,False), (85,99,[],False,False), + (85,20,[],False,False), + (85,20,[{'value':20,'time':BASE+i*5} for i in range(1,10)],False,False), + (85,20,[{'value':20,'time':BASE-121-i} for i in range(12)],False,False), + (85,99,None,False,False), + (85,float('nan'),None,False,False), (85,float('inf'),None,False,False), + (85,True,None,False,False), (85,10**400,None,False,False)): + with self.subTest(warning=warning,current=str(current),legacy=legacy), case() as store: + if legacy: first = record(store) + else: + Clock.epoch=BASE-600 + cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)],warning=warning) + first=sql(store,'SELECT first_seen FROM errors')[0][0] + Clock.epoch=BASE; cpu(store,current,history,warning=warning) + self.assertEqual(bool(store.get_recovery_evidence('cpu_usage',first)),expected) + + def test_service_unavailable_removed_overall_ok_and_ack_controls(self): + from notification_recovery_fixture import service, record + for rc,stdout,raised in ((3,'inactive\n',False),(4,'unknown\n',False),(0,'active extra\n',False),(1,'active\n',False),(0,'',True)): + with self.subTest(rc=rc,stdout=stdout,raised=raised),case() as store: + Clock.epoch=BASE-600; service(store,3,'inactive\n') + first=sql(store,'SELECT first_seen FROM errors')[0][0] + Clock.epoch=BASE; service(store,rc,stdout,raised) + self.assertIsNone(store.get_recovery_evidence('pve_service_pvedaemon',first)) + self.assertFalse(poll(store,first,'pve_service_pvedaemon','pve_services')[0]) + for removed in ('pvedaemon','corosync'): + with self.subTest(removed=removed),case() as store: + first=record(store,'pve_service_'+removed,'pve_services','service inactive',{'service':removed}) + result,calls=service(store,services=(),clustered=False) + self.assertEqual(result['status'],'OK'); self.assertEqual(calls,[]) + store.clear_error('pve_service_'+removed) + self.assertFalse(poll(store,first,'pve_service_'+removed,'pve_services')[0][0]['is_recovery']) + with case() as store: + first=record(store,'pve_service_corosync','pve_services','corosync inactive',{'service':'corosync'}) + result,calls=service(store,services=('pvedaemon',),clustered=False) + self.assertEqual(result['status'],'OK') + self.assertEqual([c[0][-1] for c in calls],['pvedaemon']) + self.assertTrue(store.is_error_active('pve_service_corosync')) + self.assertIsNone(store.get_recovery_evidence('pve_service_corosync',first)) + with case() as store: + first=record(store,'pve_service_pvedaemon','pve_services','service inactive') + store.acknowledge_error('pve_service_pvedaemon',suppression_hours=-1) + store.clear_error('pve_service_pvedaemon',check_evidence={'check':'pve_service_pvedaemon','checked_at':BASE,'service':'pvedaemon','state':'active','returncode':0}) + self.assertEqual(sql(store,'SELECT id FROM errors'),[]) + self.assertIsNone(store.get_recovery_evidence('pve_service_pvedaemon',first)) + + def test_native_binding_latest_closure_rollbacks_and_consistent_ack_read(self): + import contextlib, json, sqlite3 + for mutation in ('row_id','first_seen','closure','last_seen','latest_clear','latest_resolve','ack','event_insert_failure','ack_before_join'): + with self.subTest(mutation=mutation),case() as store: + Clock.epoch=BASE-600 + cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)]) + first=sql(store,'SELECT first_seen FROM errors')[0][0] + Clock.epoch=BASE + if mutation=='event_insert_failure': + sql(store,"CREATE TRIGGER fail_resolve BEFORE INSERT ON events WHEN NEW.event_type='resolved' BEGIN SELECT RAISE(ABORT,'fixture'); END") + self.assertEqual(cpu(store)['status'],'UNKNOWN') + self.assertIsNone(sql(store,'SELECT resolved_at FROM errors')[0][0]) + continue + cpu(store); self.assertTrue(store.get_recovery_evidence('cpu_usage',first)) + if mutation in ('row_id','first_seen','closure'): + data=json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0]) + field={'row_id':'id','first_seen':'first_seen','closure':'resolved_at'}[mutation] + data['incident'][field]='wrong' + sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps(data),)) + elif mutation=='last_seen': sql(store,'UPDATE errors SET last_seen=?',(Clock.fromtimestamp(BASE+1).isoformat(),)) + elif mutation.startswith('latest_'): + sql(store,"INSERT INTO events(event_type,error_key,timestamp,data) VALUES(?,'cpu_usage',?,'{}')",('cleared' if mutation=='latest_clear' else 'resolved',Clock.now().isoformat())) + elif mutation=='ack': store.acknowledge_error('cpu_usage',suppression_hours=-1) + else: + original=store._db_connection; triggered=[] + class Proxy: + def __init__(self,connection):self.connection=connection + def __getattr__(self,name):return getattr(self.connection,name) + def execute(self,query,args=()): + if 'FROM errors e JOIN events' in query and not triggered: + triggered.append(True) + store.acknowledge_error('cpu_usage',suppression_hours=-1) + return self.connection.execute(query,args) + @contextlib.contextmanager + def interleaved(**kw): + with original(**kw) as conn: yield Proxy(conn) + store._db_connection=interleaved + self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) + if mutation=='ack_before_join': self.assertTrue(triggered) + with case() as store: + sql(store,"INSERT INTO errors(error_key,category,severity,reason,first_seen,last_seen) VALUES('cpu_usage','cpu','WARNING','fixture','x','x')") + with self.assertRaises(sqlite3.IntegrityError): + sql(store,"INSERT INTO errors(error_key,category,severity,reason,first_seen,last_seen) VALUES('cpu_usage','cpu','WARNING','duplicate','x','x')") + + def test_malformed_history_declines_proof_without_changing_operational_clear(self): + with case() as store: + Clock.epoch=BASE-600 + cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)]) + first=sql(store,'SELECT first_seen FROM errors')[0][0] + Clock.epoch=BASE + result=cpu(store,20,[{'value':10**400,'time':BASE-i*5} for i in range(1,10)]) + self.assertEqual(result['status'],'OK') + self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) + +if __name__ == '__main__': unittest.main() diff --git a/.github/scripts/tests/test_notification_recovery_evidence.py b/.github/scripts/tests/test_notification_recovery_evidence.py index 45f316f8..096feed2 100644 --- a/.github/scripts/tests/test_notification_recovery_evidence.py +++ b/.github/scripts/tests/test_notification_recovery_evidence.py @@ -1,100 +1,68 @@ -"""Fresh existing-check provenance; extracted consumers, real disposable SQLite.""" -import contextlib -import datetime +"""Fresh existing-check provenance; native initializer and disposable SQLite.""" import json -import os -import sqlite3 -import tempfile import time -import types import unittest from unittest.mock import patch -from notification_fixture import extract, SCRIPTS, templates, LANGUAGES +from notification_fixture import templates, LANGUAGES from notification_final_fixture import deliver +from notification_recovery_fixture import case, Clock, BASE, cpu, sql, poll + + +def original_cpu(store): + Clock.epoch = BASE-600 + result = cpu(store, 90, [{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)]) + assert result['status'] == 'WARNING' + Clock.epoch = BASE + return sql(store, 'SELECT first_seen FROM errors')[0][0] + class RecoveryEvidenceTests(unittest.TestCase): def test_cpu_success_provenance_is_persisted_only_after_normal_samples(self): - events=[] - with tempfile.TemporaryDirectory() as scratch: - db=scratch+'/health.sqlite' - conn=sqlite3.connect(db) - conn.execute('CREATE TABLE errors(id INTEGER PRIMARY KEY,error_key TEXT,details TEXT,resolved_at TEXT,resolution_type TEXT,resolution_reason TEXT)') - conn.execute("INSERT INTO errors(error_key,details) VALUES ('cpu_usage','{}')") - conn.commit();conn.close() - @contextlib.contextmanager - def connection(): - c=sqlite3.connect(db) - try:yield c - finally:c.close() - ns={'datetime':datetime.datetime,'json':json} - resolve=extract(SCRIPTS/'health_persistence.py','_resolve_error_impl','HealthPersistence',ns) - store=types.SimpleNamespace(_db_connection=connection,_entity_from_details=lambda details:'',_record_event=lambda cursor,kind,key,data:events.append(data)) - store.resolve_error=lambda key,reason,**kw:resolve(store,key,reason,**kw) - ns={'Dict':dict,'Any':object,'os':os,'time':time,'health_persistence':store,'psutil':types.SimpleNamespace(cpu_percent=lambda **kw:20,cpu_count=lambda:4)} - check=extract(SCRIPTS/'health_monitor.py','_check_cpu_with_hysteresis','HealthMonitor',ns) - target=types.SimpleNamespace(state_history={'cpu_usage':[{'value':20,'time':time.time()-i*10} for i in range(10)]},CPU_CRITICAL=95,CPU_WARNING=85,CPU_RECOVERY=75,CPU_CRITICAL_DURATION=300,CPU_WARNING_DURATION=300,CPU_RECOVERY_DURATION=120,_check_cpu_temperature=lambda:None) - result=check(target) - self.assertEqual(result['status'],'OK') - self.assertTrue(events[-1].get('check_evidence'),events) - proof=events[-1]['check_evidence'] - self.assertEqual(proof['check'],'cpu_usage') - self.assertGreaterEqual(proof['checked_at'],time.time()-5) - # Existing generic resolve callers (cleanup/exclusion) get no proof. - conn=sqlite3.connect(db);conn.execute('UPDATE errors SET resolved_at=NULL');conn.commit();conn.close() - resolve(store,'cpu_usage','No longer present') - self.assertFalse(events[-1].get('check_evidence')) + with case() as store: + first = original_cpu(store) + self.assertEqual(cpu(store)['status'], 'OK') + proof = store.get_recovery_evidence('cpu_usage', first) + self.assertTrue(proof) + self.assertEqual(proof['check'], 'cpu_usage') + self.assertEqual(proof['checked_at'], BASE) + # A generic closure never gains proof; use another actual native row. + store.record_error('pve_service_test','pve_services','CRITICAL','inactive') + store.resolve_error('pve_service_test','No longer present') + self.assertFalse(json.loads(sql(store,"SELECT data FROM events ORDER BY id DESC LIMIT 1")[0][0]).get('check_evidence')) def test_recovery_query_requires_fresh_same_incident_proof(self): - with tempfile.TemporaryDirectory() as scratch: - db=scratch+'/health.sqlite' - conn=sqlite3.connect(db) - conn.execute('CREATE TABLE errors(id INTEGER PRIMARY KEY,error_key TEXT,first_seen TEXT,last_seen TEXT,resolved_at TEXT,acknowledged INTEGER)') - conn.execute('CREATE TABLE events(id INTEGER PRIMARY KEY,event_type TEXT,error_key TEXT,timestamp TEXT,data TEXT)') - now=datetime.datetime.now(); first=(now-datetime.timedelta(minutes=10)).isoformat(); last=(now-datetime.timedelta(minutes=1)).isoformat(); resolved=now.isoformat() - proof={'check':'cpu_usage','checked_at':now.timestamp()} - conn.execute('INSERT INTO errors VALUES(1,?,?,?,?,0)',('cpu_usage',first,last,resolved)) - conn.execute('INSERT INTO events VALUES(1,?,?,?,?)',('resolved','cpu_usage',resolved,json.dumps({'check_evidence':proof}))) - conn.commit();conn.close() - @contextlib.contextmanager - def connection(**kwargs): - c=sqlite3.connect(db) - try:yield c - finally:c.close() - ns={'datetime':datetime.datetime,'json':json,'time':time} - tree=(SCRIPTS/'health_persistence.py').read_text() - query=extract(SCRIPTS/'health_persistence.py','get_recovery_evidence','HealthPersistence',ns) if 'def get_recovery_evidence(' in tree else lambda *args:None - store=types.SimpleNamespace(_db_connection=connection) - self.assertEqual(query(store,'cpu_usage',first),proof) - self.assertIsNone(query(store,'cpu_usage','different incident')) - for field,value in [('acknowledged',1),('resolved_at',None),('last_seen',(now+datetime.timedelta(seconds=1)).isoformat())]: - conn=sqlite3.connect(db);conn.execute(f'UPDATE errors SET {field}=?',(value,));conn.commit();conn.close() - self.assertIsNone(query(store,'cpu_usage',first)) - conn=sqlite3.connect(db);conn.execute('UPDATE errors SET acknowledged=0,resolved_at=?,last_seen=?',(resolved,last));conn.commit();conn.close() - for bad in (None,{'check':'cpu_usage','checked_at':now.timestamp()-7201},{'check':'storage_removed','checked_at':now.timestamp()},{'check':'cpu_usage','checked_at':float('inf')},{'check':'cpu_usage','checked_at':10**400}): - conn=sqlite3.connect(db);conn.execute('UPDATE events SET data=?',(json.dumps({'check_evidence':bad}),));conn.commit();conn.close() - self.assertIsNone(query(store,'cpu_usage',first)) + with case() as store: + first = original_cpu(store); cpu(store) + proof = store.get_recovery_evidence('cpu_usage',first) + self.assertTrue(proof) + self.assertIsNone(store.get_recovery_evidence('cpu_usage','different incident')) + saved = sql(store,'SELECT last_seen,resolved_at FROM errors')[0] + for field,value in [('acknowledged',1),('resolved_at',None),('last_seen',Clock.fromtimestamp(BASE+1).isoformat())]: + sql(store,f'UPDATE errors SET {field}=?',(value,)) + self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) + sql(store,'UPDATE errors SET acknowledged=0,last_seen=?,resolved_at=?',saved) + data = json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0]) + for bad in (None,{'check':'cpu_usage','checked_at':BASE-7201}, {'check':'storage_removed','checked_at':BASE}, {'check':'cpu_usage','checked_at':float('inf')}, {'check':'cpu_usage','checked_at':10**400}): + sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps({**data,'check_evidence':bad}),)) + self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) def test_poller_and_all_consumers_distinguish_proven_recovery_from_disappearance(self): - import sys - ns={'time':time,'json':json,'Dict':dict,'NotificationEvent':lambda *a,**kw:types.SimpleNamespace(event_type=a[0],severity=a[1],data=a[2])} - poll=extract(SCRIPTS/'notification_events.py','_check_persistent_health','PollingCollector',ns) - for proof in (None,{'check':'cpu_usage','checked_at':time.time()}): - events=[] - store=types.SimpleNamespace(get_active_errors=lambda:[],is_error_acknowledged=lambda key:False,get_recovery_evidence=lambda *a:proof) - collector=types.SimpleNamespace(_hostname='node-a',_ENTITY_MAP={'cpu':('node','')},_first_poll_done=True,_known_errors={'cpu_usage':{'category':'cpu','reason':'CPU high','severity':'WARNING','first_seen':'2026-09-30T00:00:00'}},_notified_severity={'cpu_usage':'WARNING'},_last_notified={'cpu_usage':1},_queue=types.SimpleNamespace(put=events.append),_guest_storage_error_is_now_foreign=lambda *a:False,_save_known_errors_meta=lambda:None) - with patch.dict(sys.modules,{'health_persistence':types.SimpleNamespace(health_persistence=store)}):poll(collector) - self.assertEqual(len(events),1) - event=events[0] - self.assertEqual(event.data.get('recovery_outcome'),'resolved' if proof else 'no_longer_reported') - self.assertEqual(event.data['is_recovery'],bool(proof)) - for lang in LANGUAGES: - for manual in (False,True): - result=deliver(event.event_type,event.data,event.severity,lang,manual=manual) - if proof: - self.assertIn(templates.runtime_message('healthRecovery.title',lang,hostname='node-a',category='cpu',entity_suffix=''),result['title']) - self.assertIn('background:#f0fdf4;',result['html']) - else:self.assertNotIn('background:#f0fdf4;',result['html']) - self.assertEqual(result['text'].count(event.data['reason']),1) + for proved in (False,True): + with case() as store: + first = original_cpu(store) + if proved: cpu(store) + else: store.resolve_error('cpu_usage','No longer present') + data = poll(store,first,reason='CPU high')[0][0] + self.assertEqual(data['is_recovery'],proved) + self.assertEqual(data['recovery_outcome'],'resolved' if proved else 'no_longer_reported') + with patch('health_recovery.time.time',return_value=BASE): + for lang in LANGUAGES: + for manual in (False,True): + result = deliver('error_resolved',data,'OK',lang,manual=manual) + self.assertEqual('background:#f0fdf4;' in result['html'],proved) + if proved: + self.assertIn(templates.runtime_message('healthRecovery.title',lang,hostname='node-a',category='cpu',entity_suffix=''),result['title']) + self.assertEqual(result['text'].count(data['reason']),1) def test_manual_recovery_flag_alone_is_not_authoritative_evidence(self): data={'hostname':'node-a','category':'cpu','reason':'Observation disappeared','duration':'1h','original_severity':'WARNING','recovery_outcome':'resolved'} diff --git a/.github/scripts/tests/test_notification_review_corrections.py b/.github/scripts/tests/test_notification_review_corrections.py new file mode 100644 index 00000000..3928a203 --- /dev/null +++ b/.github/scripts/tests/test_notification_review_corrections.py @@ -0,0 +1,24 @@ +"""Review regressions at actual consumer seams; no operational imports.""" +import unittest +from notification_fixture import receive, LANGUAGES +from notification_final_fixture import deliver + +class ReviewCorrectionTests(unittest.TestCase): + def test_subject_equivalent_late_error_is_reserved_before_cap(self): + cause = 'job-end hook denied' + # Frozen native Perl notifier/log-reader output; Rust table source-modeled. + raw = '\nDetails\n=======\nVMID Name Status Time Size Filename \n100 web err 1m 1s 0 B null \n\nTotal running time: 1m 1s\nTotal size: 0 B\n\nLogs\n====\nvzdump --all 1 --storage PBS --mode snapshot\n\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 0\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 1\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 2\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 3\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 4\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 5\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 6\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 7\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 8\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 9\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 10\n100: 2026-09-29 17:00:00 ERROR: earlier diagnostic 11\n100: 2026-09-29 17:00:00 ERROR: job-end hook denied\n\n\n' + event = receive(raw, 'error', 'vzdump backup status (raw-host): backup failed: ' + cause) + for language in LANGUAGES: + for manual in (False, True): + with self.subTest(language=language, manual=manual): + result = deliver(event.event_type, {**event.data, 'hostname':'alias {rack.location}'}, event.severity, language, manual=manual) + self.assertEqual(result['text'].count(cause), 1) + self.assertNotIn('raw-host', result['text']) + diagnostics = [line for line in result['body'].splitlines() if 'ERROR:' in line] + self.assertLessEqual(len(diagnostics), 8) + self.assertLessEqual(len('\n'.join(diagnostics)), 1024) + self.assertTrue(all(len(line) <= 512 for line in diagnostics)) + self.assertEqual(result['data']['pve_message'], raw) + +if __name__ == '__main__': unittest.main() diff --git a/AppImage/scripts/health_monitor.py b/AppImage/scripts/health_monitor.py index 565f68e3..fdfe931f 100644 --- a/AppImage/scripts/health_monitor.py +++ b/AppImage/scripts/health_monitor.py @@ -1427,6 +1427,8 @@ class HealthMonitor: 'details': f'Sustained for {actual_duration}s above {self.CPU_CRITICAL}%.', 'cpu_percent': cpu_percent, 'duration': actual_duration, + 'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL, + 'recovery': self.CPU_RECOVERY}, }, ) elif len(warning_samples) >= WARNING_MIN_SAMPLES and len(recovery_samples) < RECOVERY_MIN_SAMPLES: @@ -1446,6 +1448,8 @@ class HealthMonitor: 'details': f'Sustained for {actual_duration}s above {self.CPU_WARNING}%.', 'cpu_percent': cpu_percent, 'duration': actual_duration, + 'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL, + 'recovery': self.CPU_RECOVERY}, }, ) else: @@ -1453,8 +1457,21 @@ class HealthMonitor: reason = None # CPU is normal - auto-resolve any existing CPU errors evidence = None - if cpu_percent < self.CPU_RECOVERY and len(recovery_samples) >= RECOVERY_MIN_SAMPLES: - evidence = {'check': 'cpu_usage', 'checked_at': current_time} + # Presentation proof is stricter than operational hysteresis: + # a supported warning can be below the fixed recovery cutoff. + from health_recovery import _finite_number + criterion = min(self.CPU_WARNING, self.CPU_RECOVERY) + normal_samples = [entry for entry in self.state_history[state_key] + if _finite_number(entry['value']) and 0 <= entry['value'] < criterion + and _finite_number(entry['time']) + and 0 <= current_time - entry['time'] <= self.CPU_RECOVERY_DURATION] + if (_finite_number(cpu_percent) and 0 <= cpu_percent < criterion + and len(normal_samples) >= RECOVERY_MIN_SAMPLES): + evidence = {'check': 'cpu_usage', 'checked_at': current_time, + 'value': cpu_percent, 'normal_samples': len(normal_samples), + 'max_sample': max(entry['value'] for entry in normal_samples), + 'policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL, + 'recovery': self.CPU_RECOVERY}} health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal', check_evidence=evidence) @@ -3904,6 +3921,7 @@ class HealthMonitor: failed_services = [] service_details = {} + active_evidence = {} for service in services_to_check: try: @@ -3918,6 +3936,10 @@ class HealthMonitor: if result.returncode != 0 or status != 'active': failed_services.append(service) service_details[service] = status or 'inactive' + else: + active_evidence[service] = {'check': f'pve_service_{service}', + 'checked_at': time.time(), 'service': service, 'state': status, + 'returncode': result.returncode} except Exception: failed_services.append(service) service_details[service] = 'error' @@ -3932,7 +3954,7 @@ class HealthMonitor: if svc not in failed_services: error_key = f'pve_service_{svc}' if health_persistence.is_error_active(error_key): - health_persistence.clear_error(error_key) + health_persistence.clear_error(error_key, check_evidence=active_evidence.get(svc)) # Build checks dict with status per service checks = {} diff --git a/AppImage/scripts/health_persistence.py b/AppImage/scripts/health_persistence.py index 35458418..8c108c13 100644 --- a/AppImage/scripts/health_persistence.py +++ b/AppImage/scripts/health_persistence.py @@ -600,7 +600,7 @@ class HealthPersistence: cursor.execute(''' SELECT id, acknowledged, resolved_at, category, severity, first_seen, - notification_sent, suppression_hours, acknowledged_at + notification_sent, suppression_hours, acknowledged_at, details FROM errors WHERE error_key = ? ''', (error_key,)) existing = cursor.fetchone() @@ -609,7 +609,7 @@ class HealthPersistence: if existing: (err_id, ack, resolved_at, old_cat, old_severity, first_seen, - notif_sent, stored_suppression, acknowledged_at) = existing + notif_sent, stored_suppression, acknowledged_at, old_details_json) = existing if ack == 1: # SAFETY OVERRIDE: Critical CPU temperature ALWAYS re-triggers @@ -680,6 +680,18 @@ class HealthPersistence: conn.commit() return event_info + # Original CPU policy is immutable for this row's incident. + # Never upgrade a legacy row from later/current settings. + if error_key == 'cpu_usage': + try: + old_details = json.loads(old_details_json or '{}') + except (ValueError, TypeError): + old_details = {} + details = dict(details) if isinstance(details, dict) else {} + details.pop('cpu_policy', None) + if isinstance(old_details, dict) and 'cpu_policy' in old_details: + details['cpu_policy'] = old_details['cpu_policy'] + details_json = json.dumps(details) # Not acknowledged - update existing active error cursor.execute(''' UPDATE errors @@ -776,7 +788,7 @@ class HealthPersistence: # was created — otherwise "Storage 'Tuxis' unavailable" # comes back as "Resolved - Storage" with no identity. cursor.execute( - 'SELECT details FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1', + 'SELECT details, id, first_seen, resolved_at FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1', (error_key,), ) row = cursor.fetchone() @@ -786,11 +798,18 @@ class HealthPersistence: stored_details = json.loads(row[0]) except Exception: stored_details = None + # Legacy/mismatched original policy cannot certify normality. + if error_key == 'cpu_usage' and (not isinstance(stored_details, dict) + or not isinstance(check_evidence, dict) + or check_evidence.get('policy') != stored_details.get('cpu_policy') + or not stored_details.get('cpu_policy')): + check_evidence = None self._record_event(cursor, 'resolved', error_key, { 'reason': reason, # Only explicit current-check callers attach this proof. # Generic resolve/cleanup remains neutral. 'check_evidence': check_evidence, + 'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': row[3]}, 'entity': self._entity_from_details(stored_details), 'details': stored_details or {}, }) @@ -800,41 +819,40 @@ class HealthPersistence: def get_recovery_evidence(self, error_key: str, first_seen: str): """Return fresh same-incident native check proof, never absence of errors. - Initially only the host CPU check has a stable condition identity. Other + Host CPU and exact per-service active checks carry provenance. Other checks, generic clears, excluded/deleted records and legacy events stay neutral until they have equivalent per-condition provenance. """ - if error_key != 'cpu_usage' or not first_seen: + if not isinstance(error_key, str) or not first_seen or not ( + error_key == 'cpu_usage' or error_key.startswith('pve_service_')): return None try: - with self._db_connection() as conn: + # One SQLite statement is one consistent row/ack/closure snapshot. + # Latest closure by event id, never search past a generic clear. + with self._db_lock, self._db_connection() as conn: row = conn.execute(''' - SELECT first_seen, last_seen, resolved_at, acknowledged - FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1 + SELECT e.first_seen, e.last_seen, e.resolved_at, e.acknowledged, + e.id, v.timestamp, v.data + FROM errors e JOIN events v ON v.id = ( + SELECT id FROM events WHERE error_key = e.error_key + AND event_type IN ('resolved', 'cleared') ORDER BY id DESC LIMIT 1 + ) WHERE e.error_key = ? ''', (error_key,)).fetchone() - if not row or row[0] != first_seen or not row[2] or row[3]: - return None - event = conn.execute(''' - SELECT timestamp, data FROM events - WHERE error_key = ? AND event_type = 'resolved' - ORDER BY id DESC LIMIT 1 - ''', (error_key,)).fetchone() - if not event: + if not row or row[0] != first_seen or not row[2] or row[3]: return None - proof = json.loads(event[1]).get('check_evidence') - if not isinstance(proof, dict) or proof.get('check') != error_key: + event_data = json.loads(row[6]) + if event_data.get('incident') != {'id': row[4], 'first_seen': row[0], 'resolved_at': row[2]}: return None - checked = proof.get('checked_at') - if not isinstance(checked, (int, float)) or isinstance(checked, bool): + proof = event_data.get('check_evidence') + from health_recovery import valid_check_evidence + if not valid_check_evidence(error_key, proof, now=datetime.now().timestamp()): return None - checked = float(checked) - # Reuse the collector's existing two-hour freshness boundary. - now = datetime.now().timestamp() - if not 0 <= now - checked <= 7200: + if error_key == 'cpu_usage' and proof.get('policy') != event_data.get('details', {}).get('cpu_policy'): return None + checked = float(proof['checked_at']) last_seen = datetime.fromisoformat(row[1]).timestamp() resolved = datetime.fromisoformat(row[2]).timestamp() - recorded = datetime.fromisoformat(event[0]).timestamp() + recorded = datetime.fromisoformat(row[5]).timestamp() if not last_seen <= checked <= resolved <= recorded: return None return proof @@ -906,7 +924,7 @@ class HealthPersistence: return False - def clear_error(self, error_key: str): + def clear_error(self, error_key: str, *, check_evidence=None): """ Remove/resolve a specific error immediately. Used when the condition that caused the error no longer exists @@ -928,7 +946,7 @@ class HealthPersistence: # Check if this error was acknowledged (dismissed) cursor.execute(''' - SELECT acknowledged FROM errors WHERE error_key = ? + SELECT acknowledged, id, first_seen FROM errors WHERE error_key = ? ''', (error_key,)) row = cursor.fetchone() @@ -947,7 +965,10 @@ class HealthPersistence: ''', (now, error_key)) if cursor.rowcount > 0: - self._record_event(cursor, 'cleared', error_key, {'reason': 'condition_resolved'}) + self._record_event(cursor, 'cleared', error_key, { + 'reason': 'condition_resolved', 'check_evidence': check_evidence, + 'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': now}, + }) conn.commit() diff --git a/AppImage/scripts/health_recovery.py b/AppImage/scripts/health_recovery.py new file mode 100644 index 00000000..9a8a00df --- /dev/null +++ b/AppImage/scripts/health_recovery.py @@ -0,0 +1,51 @@ +"""Bounded recovery metadata admission, without importing monitor singletons. + +Native persistence additionally binds the exact incident and closure. Manual +notifications remain authenticated caller assertions: shape validation cannot +establish that an asserted measurement actually happened. +""" +import math +import time +from typing import TypeGuard + + +def _finite_number(value) -> TypeGuard[int | float]: + try: + return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value) + except (OverflowError, ValueError, TypeError): + return False + + +def valid_check_evidence(error_key, proof, *, now=None): + """Validate a supported measurement contract, not its external authenticity.""" + if not isinstance(proof, dict) or proof.get('check') != error_key: + return False + checked = proof.get('checked_at') + now = time.time() if now is None else now + if not _finite_number(checked) or not _finite_number(now) or not 0 <= now-checked <= 7200: + return False + if error_key == 'cpu_usage': + policy = proof.get('policy') + if not isinstance(policy, dict) or set(policy) != {'warning', 'critical', 'recovery'}: + return False + if not all(_finite_number(v) and 1 <= v <= 100 for v in policy.values()): + return False + if policy['warning'] > policy['critical']: + return False + value, maximum, count = proof.get('value'), proof.get('max_sample'), proof.get('normal_samples') + return (_finite_number(value) and _finite_number(maximum) + and 0 <= value <= maximum < min(policy['warning'], policy['recovery']) + and isinstance(count, int) and not isinstance(count, bool) and count >= 10) + if isinstance(error_key, str) and error_key.startswith('pve_service_'): + service = error_key[len('pve_service_'):] + return (bool(service) and proof.get('service') == service + and proof.get('state') == 'active' + and type(proof.get('returncode')) is int and proof['returncode'] == 0) + return False + + +def presents_recovery(data): + """One presentation predicate shared by template, icon and email badge.""" + return (isinstance(data, dict) and data.get('recovery_outcome') == 'resolved' + and data.get('is_recovery') is True + and valid_check_evidence(data.get('error_key'), data.get('check_evidence'))) diff --git a/AppImage/scripts/notification_channels.py b/AppImage/scripts/notification_channels.py index b0b106ab..584ad7a1 100644 --- a/AppImage/scripts/notification_channels.py +++ b/AppImage/scripts/notification_channels.py @@ -1038,10 +1038,8 @@ class EmailChannel(NotificationChannel): # Determine group for section header event_type = data.get('_event_type', '') if event_type == 'error_resolved': - if (data.get('recovery_outcome') == 'resolved' - and data.get('is_recovery') is True - and isinstance(data.get('check_evidence'), dict) - and data['check_evidence'].get('check') == 'cpu_usage'): + from health_recovery import presents_recovery + if presents_recovery(data): sev.update(self._SEV_STYLE['OK']) sev['label'] = _runtime_notification_text('healthRecovery.status', data) else: @@ -1070,8 +1068,11 @@ class EmailChannel(NotificationChannel): or backup_email or data.get('_restore_summary') or data.get('_backup_summary')) temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else '' temp_table_layout = 'table-layout:fixed;' if wrap_body else '' - backup_title_wrap = temp_cell_wrap if backup_email else '' - backup_metadata_layout = 'table-layout:fixed;' if backup_email else '' + # Recovery exposes the same literal host context as backup notices. + # Keep wrapping event-scoped; unrelated mail remains byte-identical. + context_email = backup_email or event_type == 'error_resolved' + backup_title_wrap = temp_cell_wrap if context_email else '' + backup_metadata_layout = 'table-layout:fixed;' if context_email else '' section_label = _runtime_text(f'email.groups.{group}', data) report_label = _runtime_text('email.report', data, group=section_label) host_label = _runtime_text('email.host', data) diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index 99ae05ff..5cc9f2b6 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -2070,10 +2070,8 @@ def render_template(event_type: str, data: Dict[str, Any], return '' safe_vars = _SafeDict(variables) - if (event_type == 'error_resolved' and data.get('recovery_outcome') == 'resolved' - and data.get('is_recovery') is True - and isinstance(data.get('check_evidence'), dict) - and data['check_evidence'].get('check') == 'cpu_usage'): + from health_recovery import presents_recovery + if event_type == 'error_resolved' and presents_recovery(data): safe_vars['_health_title'] = runtime_message('healthRecovery.title', language, **variables) safe_vars['_health_body'] = runtime_message('healthRecovery.body', language, **variables) template['title'] = '{_health_title}' @@ -2089,13 +2087,14 @@ def render_template(event_type: str, data: Dict[str, Any], # parse the table/logs and format a rich body instead of the sparse template. pve_message = data.get('pve_message', '') backup_diagnostics = [] + principal_cause = None - def bounded_backup_diagnostics(lines): + def bounded_backup_diagnostics(lines, principal_cause=None): # 1024 chars matches the repository's small-channel message convention; # 8 lines keeps repeated producer warnings readable. Inventory/title # size is separate: this is not a one-Telegram-message guarantee. unique = list(dict.fromkeys(line for line in lines if line.strip())) - principal = next((line for line in unique if re.search(r'\b(?:ERROR:|TASK ERROR:)', line, re.IGNORECASE)), None) + principal = principal_cause or next((line for line in unique if re.search(r'\b(?:ERROR:|TASK ERROR:)', line, re.IGNORECASE)), None) if principal: unique.remove(principal) unique.insert(0, principal) @@ -2189,16 +2188,18 @@ def render_template(event_type: str, data: Dict[str, Any], source_subject = cause.group(1).strip() if cause else '' if source_subject.lower() == 'multiple problems': source_subject = '' - cause_in_diagnostics = any( - line.strip() == source_subject or - re.split(r'\b(?:TASK ERROR:|ERROR:)\s*', line, maxsplit=1, flags=re.IGNORECASE)[-1].strip() == source_subject - for line in backup_diagnostics) - if source_subject and source_subject not in body_text and not cause_in_diagnostics: - # A unique job/setup cause must survive a warning-heavy report. - backup_diagnostics.insert(0, source_subject) + if source_subject and source_subject not in body_text: + # Reserve the subject-equivalent diagnostic BEFORE the cap. Finding + # it in uncapped logs is not enough: that late line could be omitted. + principal_cause = next((line for line in backup_diagnostics + if line.strip() == source_subject or + re.split(r'\b(?:TASK ERROR:|ERROR:)\s*', line, maxsplit=1, flags=re.IGNORECASE)[-1].strip() == source_subject), None) + if not principal_cause: + principal_cause = source_subject + backup_diagnostics.insert(0, source_subject) if backup_diagnostics: - body_text += '\n' + bounded_backup_diagnostics(backup_diagnostics) + body_text += '\n' + bounded_backup_diagnostics(backup_diagnostics, principal_cause) # Clean up: collapse runs of 3+ blank lines into 1, remove trailing whitespace import re as _re @@ -2531,10 +2532,8 @@ def enrich_with_emojis(event_type: str, title: str, body: str, severity = data.get('severity', 'INFO') icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '') - if (event_type == 'error_resolved' and data.get('recovery_outcome') == 'resolved' - and data.get('is_recovery') is True - and isinstance(data.get('check_evidence'), dict) - and data['check_evidence'].get('check') == 'cpu_usage'): + from health_recovery import presents_recovery + if event_type == 'error_resolved' and presents_recovery(data): icon = '✅' if event_type == 'backup_complete': icon = { From 54eef11334930efb85a28eff204c4eda9b59e89e Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Thu, 1 Oct 2026 00:35:49 +0200 Subject: [PATCH 11/14] Load bounded recovery predicate only for recovery presentation --- AppImage/scripts/notification_templates.py | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index 5cc9f2b6..ebfbd364 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -2070,12 +2070,13 @@ def render_template(event_type: str, data: Dict[str, Any], return '' safe_vars = _SafeDict(variables) - from health_recovery import presents_recovery - if event_type == 'error_resolved' and presents_recovery(data): - safe_vars['_health_title'] = runtime_message('healthRecovery.title', language, **variables) - safe_vars['_health_body'] = runtime_message('healthRecovery.body', language, **variables) - template['title'] = '{_health_title}' - template['body'] = '{_health_body}' + if event_type == 'error_resolved': + from health_recovery import presents_recovery + if presents_recovery(data): + safe_vars['_health_title'] = runtime_message('healthRecovery.title', language, **variables) + safe_vars['_health_body'] = runtime_message('healthRecovery.body', language, **variables) + template['title'] = '{_health_title}' + template['body'] = '{_health_body}' try: title = template['title'].format_map(safe_vars) except (ValueError, IndexError): @@ -2532,9 +2533,10 @@ def enrich_with_emojis(event_type: str, title: str, body: str, severity = data.get('severity', 'INFO') icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '') - from health_recovery import presents_recovery - if event_type == 'error_resolved' and presents_recovery(data): - icon = '✅' + if event_type == 'error_resolved': + from health_recovery import presents_recovery + if presents_recovery(data): + icon = '✅' if event_type == 'backup_complete': icon = { 'confirmed': '💾✅', 'completed_with_warnings': '💾⚠️', 'failed': '💾❌', From aac75a0753588bfc68c68006022661b794ee8c6e Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Thu, 1 Oct 2026 09:03:27 +0200 Subject: [PATCH 12/14] fix(notifications): ship recovery helper and reject superseded proofs --- .../test_notification_packaged_recovery.py | 83 +++++++++ .../test_notification_principal_inventory.py | 50 ++++++ .../tests/test_notification_recovery_order.py | 170 ++++++++++++++++++ AppImage/scripts/build_appimage.sh | 1 + AppImage/scripts/health_persistence.py | 8 +- AppImage/scripts/notification_templates.py | 2 +- 6 files changed, 311 insertions(+), 3 deletions(-) create mode 100644 .github/scripts/tests/test_notification_packaged_recovery.py create mode 100644 .github/scripts/tests/test_notification_principal_inventory.py create mode 100644 .github/scripts/tests/test_notification_recovery_order.py diff --git a/.github/scripts/tests/test_notification_packaged_recovery.py b/.github/scripts/tests/test_notification_packaged_recovery.py new file mode 100644 index 00000000..0c359634 --- /dev/null +++ b/.github/scripts/tests/test_notification_packaged_recovery.py @@ -0,0 +1,83 @@ +"""Execute the build's literal backend copies, then a clean shipped-only runtime.""" +import os +from pathlib import Path +import re +import subprocess +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[3] +PROBE = r''' +import ast, pathlib, sys, types, typing +stage = pathlib.Path(sys.argv[1]) +sys.path.insert(0, str(stage)) +assert 'health_recovery' not in sys.modules +import notification_templates as templates +from notification_channels import EmailChannel +import health_recovery +for module in (templates, health_recovery, sys.modules['notification_channels']): + assert pathlib.Path(module.__file__).parent == stage +assert not any('projects/proxmox' in p for p in sys.path) +templates._get_hostname = lambda: 'node-a' +channel = object.__new__(EmailChannel) +channel.subject_prefix = '[ProxMenux]' +neutral = {'hostname': 'node-a', 'category': 'cpu', 'reason': 'CPU high', '_event_type': 'error_resolved'} +templates.render_template('error_resolved', neutral) +templates.enrich_with_emojis('error_resolved', 'Observation', 'Body', neutral) +channel._format_html('Observation', 'Body', 'OK', neutral) +templates.render_template('node_reconnect', {'hostname': 'node-a'}) +now = 1000.0 +calls = [] +ns = dict(vars(typing), time=types.SimpleNamespace(time=lambda: now), + os=types.SimpleNamespace(cpu_count=lambda: 4), + psutil=types.SimpleNamespace(cpu_percent=lambda **kw: 20, cpu_count=lambda: 4), + health_persistence=types.SimpleNamespace(resolve_error=lambda *a, **kw: calls.append((a, kw)))) +tree = ast.parse((stage/'health_monitor.py').read_text()) +owner = next(n for n in tree.body if isinstance(n, ast.ClassDef) and n.name == 'HealthMonitor') +node = next(n for n in owner.body if isinstance(n, ast.FunctionDef) and n.name == '_check_cpu_with_hysteresis') +exec(compile(ast.Module(body=[node], type_ignores=[]), 'shipped_cpu', 'exec'), ns) +monitor = types.SimpleNamespace(state_history={'cpu_usage': [{'value':20,'time':now-i*10} for i in range(1,11)]}, + CPU_WARNING=85, CPU_CRITICAL=95, CPU_RECOVERY=75, CPU_WARNING_DURATION=300, + CPU_CRITICAL_DURATION=300, CPU_RECOVERY_DURATION=120, _check_cpu_temperature=lambda:None) +assert ns['_check_cpu_with_hysteresis'](monitor)['status'] == 'OK' +assert len(calls) == 1 and calls[0][1]['check_evidence']['value'] == 20 +proof = calls[0][1]['check_evidence'] +native = dict(neutral, error_key='cpu_usage', check_evidence=proof, is_recovery=True, recovery_outcome='resolved') +# Actual body/icon/email consumers from the package, no source-module fixture rescue. +import unittest.mock +with unittest.mock.patch('health_recovery.time.time', return_value=now): + result = templates.render_template('error_resolved', native) + title, body = result['title'], result['body'] + assert 'Resolved' in title and 'fresh health check' in body, (title, body, proof) + rich_title, rich_body = templates.enrich_with_emojis('error_resolved', title, body, native) + assert rich_title.startswith('✅') + assert 'background:#f0fdf4;' in channel._format_html(title, body, 'OK', native) +print('shipped-only neutral body/icon/email + CPU native measurement/proof consumers PASS') +''' + + +class PackagedRecoveryTests(unittest.TestCase): + def test_actual_copy_manifest_supports_isolated_recovery_runtime(self): + source = ROOT / 'AppImage/scripts' + build = (source / 'build_appimage.sh').read_text() + lines = [line for line in build.splitlines() + if re.match(r'^cp "\$SCRIPT_DIR/[^"/]+\.py" "\$APP_DIR/usr/bin/"', line)] + self.assertTrue(lines) + catalog_copy = re.search(r'^for locale in en de es fr it pt sk sv; do\n.*?^done$', build, re.MULTILINE | re.DOTALL) + if catalog_copy is None: + self.fail('The shipped locale-copy loop was not found in the actual build script') + lines.append(catalog_copy.group(0)) + with tempfile.TemporaryDirectory(prefix='shipped-recovery-') as directory: + stage = Path(directory) / 'usr/bin' + stage.mkdir(parents=True) + copied = subprocess.run(['/bin/bash'], input='set -e\n'+'\n'.join(lines)+'\n', text=True, + capture_output=True, env={**os.environ, 'SCRIPT_DIR': str(source), 'APPIMAGE_ROOT': str(source.parent), 'APP_DIR': directory}) + self.assertEqual(copied.returncode, 0, copied.stderr) + result = subprocess.run([sys.executable, '-I', '-B', '-c', PROBE, str(stage)], + cwd=directory, text=True, capture_output=True) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + + +if __name__ == '__main__': + unittest.main() diff --git a/.github/scripts/tests/test_notification_principal_inventory.py b/.github/scripts/tests/test_notification_principal_inventory.py new file mode 100644 index 00000000..de028952 --- /dev/null +++ b/.github/scripts/tests/test_notification_principal_inventory.py @@ -0,0 +1,50 @@ +"""Principal diagnostics must not be deduplicated against inventory substrings.""" +import unittest +from notification_fixture import receive, LANGUAGES +from notification_final_fixture import deliver + + +def noisy_report(name='ordinary-web', storage='PBS', filename='null', cause='denied'): + # Frozen native Perl notifier/log-reader shape; fixed-column table is + # source-modeled from the official plaintext renderer, not a live PVE run. + return ('\nDetails\n=======\nVMID Name Status Time Size Filename\n' + f'100 {name} err 1m 1s 0 B {filename}\n\n' + 'Total running time: 1m 1s\nTotal size: 0 B\n\nLogs\n====\n' + f'vzdump --all 1 --storage {storage} --mode snapshot\n\n' + + '\n'.join(f'100: 2026-09-29 17:00:00 ERROR: earlier diagnostic {i}' for i in range(12)) + + f'\n100: 2026-09-29 17:00:00 ERROR: {cause}\n\n') + + +class PrincipalInventoryTests(unittest.TestCase): + def assert_principal(self, raw, cause): + event = receive(raw, 'error', 'vzdump backup status (raw-host): backup failed: ' + cause) + for language in LANGUAGES: + for manual in (False, True): + with self.subTest(language=language, manual=manual): + result = deliver(event.event_type, {**event.data, 'hostname':'alias {rack.location}'}, + event.severity, language, manual=manual) + exact = [line for line in result['body'].splitlines() + if line.strip() == cause or line.rstrip().endswith('ERROR: ' + cause)] + self.assertEqual(len(exact), 1) + diagnostics = [line for line in result['body'].splitlines() if 'ERROR:' in line] + self.assertLessEqual(len(diagnostics), 8) + self.assertLessEqual(len('\n'.join(diagnostics)), 1024) + self.assertTrue(all(len(line) <= 512 for line in diagnostics)) + self.assertEqual(result['data']['pve_message'], raw) + self.assertNotIn('raw-host', result['text']) + self.assertIn('alias {rack.location}', result['text']) + self.assertEqual(result['text'].count('ERROR: ' + cause), 1) + + def test_late_principal_survives_incidental_guest_storage_and_archive(self): + for name, storage, filename in (('ordinary-web','PBS','null'), ('denied','PBS','null'), + ('ordinary-web','denied','null'), ('ordinary-web','PBS','denied.tar')): + with self.subTest(name=name, storage=storage, filename=filename): + self.assert_principal(noisy_report(name,storage,filename), 'denied') + + def test_raw_principal_braces_and_markup_survive_once(self): + cause = 'denied {rack.location} ' + self.assert_principal(noisy_report('denied', cause=cause), cause) + + +if __name__ == '__main__': + unittest.main() diff --git a/.github/scripts/tests/test_notification_recovery_order.py b/.github/scripts/tests/test_notification_recovery_order.py new file mode 100644 index 00000000..9b3e1607 --- /dev/null +++ b/.github/scripts/tests/test_notification_recovery_order.py @@ -0,0 +1,170 @@ +"""Durable native observation order, not wall-clock row identity, admits proof.""" +import datetime +import json +import re +import sys +import types +import typing +import unittest +from unittest.mock import patch +from notification_recovery_fixture import case, Clock, BASE, cpu, service, sql, extract, scripts, TIME +from notification_fixture import LANGUAGES +from notification_final_fixture import deliver + + +def collector(store): + events = [] + target = types.SimpleNamespace(_hostname='alias {rack.location}', + _ENTITY_MAP={'cpu':('node',''), 'pve_services':('node','')}, _first_poll_done=False, + _known_errors={}, _notified_severity={}, _last_notified={}, SAME_ERROR_COOLDOWN=86400, + _get_cooldown_from_db=lambda *a:BASE-1, _queue=types.SimpleNamespace(put=events.append), + _save_known_errors_meta=lambda:None) + ns = dict(vars(typing), time=TIME, json=json, re=re, + NotificationEvent=lambda *a, **kw:types.SimpleNamespace(event_type=a[0], severity=a[1], data=a[2], **kw), + startup_grace=types.SimpleNamespace(should_suppress_category=lambda *a:False)) + target._guest_storage_error_is_now_foreign = extract(scripts/'notification_events.py', + '_guest_storage_error_is_now_foreign', 'PollingCollector', ns) + poll = extract(scripts/'notification_events.py', '_check_persistent_health', 'PollingCollector', ns) + def tick(): + with patch.dict(sys.modules, {'health_persistence':types.SimpleNamespace(health_persistence=store), + 'datetime':types.SimpleNamespace(**{**vars(datetime), 'datetime':Clock})}): + poll(target) + return target, events, tick + + +def abnormal(store, kind, value=99): + if kind == 'cpu': + return cpu(store, value, [{'value':value, 'time':Clock.epoch-i*5} for i in range(1,26)]) + return service(store,3,'inactive\n')[0] + + +def normal(store, kind): + return cpu(store) if kind == 'cpu' else service(store)[0] + + +def initial_closure(store, kind): + key = 'cpu_usage' if kind == 'cpu' else 'pve_service_pvedaemon' + Clock.epoch = BASE-600 + abnormal(store,kind,90) + target, events, tick = collector(store) + tick() + assert target._first_poll_done and key in target._known_errors and not events + snapshot = json.loads(json.dumps(target._known_errors)) + Clock.epoch = BASE + assert normal(store,kind)['status'] == 'OK' + first = snapshot[key]['first_seen'] + assert store.get_recovery_evidence(key,first) + return key, first, snapshot, target, events, tick + + +class RecoveryOrderTests(unittest.TestCase): + def assert_consumers(self, data, expected): + with patch('health_recovery.time.time', return_value=Clock.epoch): + for language in LANGUAGES: + for manual in (False,True): + with self.subTest(language=language, manual=manual): + result = deliver('error_resolved',data,'OK',language,manual=manual) + self.assertEqual('background:#f0fdf4;' in result['html'],expected) + self.assertIn('alias {rack.location}',result['text']) + quiet = deliver('error_resolved',data,'OK',language,quiet=True) + self.assertEqual(len(quiet['buffered']),1) + # The existing quiet digest stores the rendered title, not + # health body/proof metadata. Assert its exact outcome label. + from notification_fixture import templates + label = templates.render_template('error_resolved',data,language)['title'].split(': ',1)[-1] + self.assertIn(label, quiet['body']) + self.assertIn(label, quiet['buffered'][0][2]) + + def assert_superseded(self, kind): + for offset in (-100,0,1): + for value in ((90,99) if kind == 'cpu' else (99,)): + with self.subTest(kind=kind, offset=offset, value=value), case() as store: + key, first, snapshot, target, events, tick = initial_closure(store,kind) + oldrow = sql(store,'SELECT id,first_seen,resolved_at FROM errors')[0] + prior = sql(store,'SELECT id,event_type FROM events ORDER BY id') + Clock.epoch = BASE+offset + renewed = abnormal(store,kind,value) + self.assertEqual(renewed['status'],'WARNING' if kind == 'cpu' and value == 90 else 'CRITICAL') + self.assertEqual(sql(store,'SELECT id,first_seen,resolved_at FROM errors')[0],oldrow) + later = sql(store,'SELECT id,event_type FROM events ORDER BY id')[-1] + self.assertGreater(later[0],prior[-1][0]) + self.assertEqual(later[1],'escalated' if kind == 'cpu' and value == 99 else 'updated') + if kind == 'cpu': + policy=json.loads(sql(store,'SELECT details FROM errors')[0][0])['cpu_policy'] + self.assertEqual(policy,{'warning':85,'critical':95,'recovery':75}) + Clock.epoch = BASE+2 + tick() + self.assertEqual(len(events),1) + data = events[0].data + self.assertFalse(data['is_recovery']) + self.assertIsNone(store.get_recovery_evidence(key,first)) + self.assert_consumers(data,False) + # Existing operations do not rearm the resolved row, so a + # later normal check cannot establish a NEW native closure. + before = sql(store,'SELECT id FROM events ORDER BY id') + Clock.epoch = BASE+3 + self.assertEqual(normal(store,kind)['status'],'OK') + self.assertEqual(sql(store,'SELECT id FROM events ORDER BY id'),before) + self.assertIsNone(store.get_recovery_evidence(key,first)) + + def test_cpu_superseded_same_row_is_neutral_at_all_consumers(self): + self.assert_superseded('cpu') + + def test_service_superseded_same_row_is_neutral_at_all_consumers(self): + self.assert_superseded('service') + + def test_fresh_closure_and_repeated_noop_clear_keep_genuine_proof(self): + for kind in ('cpu','service'): + with self.subTest(kind=kind),case() as store: + key, first, snapshot, target, events, tick = initial_closure(store,kind) + before = sql(store,'SELECT id,event_type FROM events ORDER BY id') + Clock.epoch = BASE+1 + for _ in range(3): + store.clear_error(key) + store.resolve_error(key,'generic repeat') + normal(store,kind) + self.assertEqual(sql(store,'SELECT id,event_type FROM events ORDER BY id'),before) + self.assertTrue(store.get_recovery_evidence(key,first)) + tick() + self.assertEqual(len(events),1) + self.assertTrue(events[0].data['is_recovery']) + self.assert_consumers(events[0].data,True) + tick() + self.assertEqual(len(events),1) + + def test_later_actual_generic_closure_blocks_older_proof_even_clock_rollback(self): + for kind in ('cpu','service'): + with self.subTest(kind=kind),case() as store: + key, first, *_ = initial_closure(store,kind) + Clock.epoch = BASE-100 + with store._db_connection() as conn: + store._record_event(conn.cursor(),'cleared',key,{'reason':'generic actual closure','check_evidence':None}) + conn.commit() + self.assertIsNone(store.get_recovery_evidence(key,first)) + + def test_explicit_acknowledged_row_suppresses_native_proof(self): + for kind in ('cpu','service'): + with self.subTest(kind=kind),case() as store: + key, first, snapshot, target, events, tick = initial_closure(store,kind) + store.acknowledge_error(key,suppression_hours=-1) + self.assertEqual(sql(store,'SELECT acknowledged FROM errors')[0][0],1) + self.assertIsNone(store.get_recovery_evidence(key,first)) + tick() + self.assertEqual(events,[]) + + def test_new_incarnation_abnormal_order_blocks_previous_closure(self): + for kind in ('cpu','service'): + with self.subTest(kind=kind),case() as store: + key, first, *_ = initial_closure(store,kind) + old_id = sql(store,'SELECT id FROM errors')[0][0] + store.acknowledge_error(key,suppression_hours=-1) + store.clear_error(key) + Clock.epoch = BASE-100 + abnormal(store,kind,99) + self.assertGreater(sql(store,'SELECT id FROM errors')[0][0],old_id) + self.assertEqual(sql(store,'SELECT event_type FROM events ORDER BY id DESC')[0][0],'new') + self.assertIsNone(store.get_recovery_evidence(key,first)) + + +if __name__ == '__main__': + unittest.main() diff --git a/AppImage/scripts/build_appimage.sh b/AppImage/scripts/build_appimage.sh index ccbeff2b..2e7d6a0e 100755 --- a/AppImage/scripts/build_appimage.sh +++ b/AppImage/scripts/build_appimage.sh @@ -131,6 +131,7 @@ cp "$SCRIPT_DIR/auth_manager.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ cp "$SCRIPT_DIR/jwt_middleware.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ jwt_middleware.py not found" cp "$SCRIPT_DIR/health_monitor.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_monitor.py not found" cp "$SCRIPT_DIR/health_persistence.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_persistence.py not found" +cp "$SCRIPT_DIR/health_recovery.py" "$APP_DIR/usr/bin/" cp "$SCRIPT_DIR/flask_health_routes.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ flask_health_routes.py not found" cp "$SCRIPT_DIR/flask_proxmenux_routes.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ flask_proxmenux_routes.py not found" cp "$SCRIPT_DIR/post_install_versions.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ post_install_versions.py not found" diff --git a/AppImage/scripts/health_persistence.py b/AppImage/scripts/health_persistence.py index 8c108c13..d06414e2 100644 --- a/AppImage/scripts/health_persistence.py +++ b/AppImage/scripts/health_persistence.py @@ -828,14 +828,18 @@ class HealthPersistence: return None try: # One SQLite statement is one consistent row/ack/closure snapshot. - # Latest closure by event id, never search past a generic clear. + # Latest native observation/closure by durable event id: a later + # abnormal record supersedes proof even when a resolved row is + # reused and wall-clock time moves backward. No-op clears create + # no event, so they do not invalidate a genuine closure. with self._db_lock, self._db_connection() as conn: row = conn.execute(''' SELECT e.first_seen, e.last_seen, e.resolved_at, e.acknowledged, e.id, v.timestamp, v.data FROM errors e JOIN events v ON v.id = ( SELECT id FROM events WHERE error_key = e.error_key - AND event_type IN ('resolved', 'cleared') ORDER BY id DESC LIMIT 1 + AND event_type IN ('resolved', 'cleared', 'new', 'updated', 'escalated') + ORDER BY id DESC LIMIT 1 ) WHERE e.error_key = ? ''', (error_key,)).fetchone() if not row or row[0] != first_seen or not row[2] or row[3]: diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index ebfbd364..f0d055b4 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -2189,7 +2189,7 @@ def render_template(event_type: str, data: Dict[str, Any], source_subject = cause.group(1).strip() if cause else '' if source_subject.lower() == 'multiple problems': source_subject = '' - if source_subject and source_subject not in body_text: + if source_subject and source_subject not in {line.strip() for line in body_text.splitlines()}: # Reserve the subject-equivalent diagnostic BEFORE the cap. Finding # it in uncapped logs is not enough: that late line could be omitted. principal_cause = next((line for line in backup_diagnostics From 21babcc14d1114469e98e48d975dfa52319f1c22 Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Thu, 1 Oct 2026 11:18:26 +0200 Subject: [PATCH 13/14] test(notifications): allow default temporary directory in CI --- .github/scripts/tests/notification_recovery_fixture.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/scripts/tests/notification_recovery_fixture.py b/.github/scripts/tests/notification_recovery_fixture.py index 8e4f91d6..2cb7fb6f 100644 --- a/.github/scripts/tests/notification_recovery_fixture.py +++ b/.github/scripts/tests/notification_recovery_fixture.py @@ -83,4 +83,4 @@ def service(store, rc=0, stdout='active\n', raised=False, services=('pvedaemon', @contextlib.contextmanager def case(): Clock.epoch=BASE - with tempfile.TemporaryDirectory(prefix='recovery-review-db-',dir=os.environ['TMPDIR']) as d:yield make_store(d) + with tempfile.TemporaryDirectory(prefix='recovery-review-db-',dir=os.environ.get('TMPDIR')) as d:yield make_store(d) From ea5fc041b840c161d84c6542a1e2a7c4aa9cf7bd Mon Sep 17 00:00:00 2001 From: martino <32328813+f3rs3n@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:04:08 +0200 Subject: [PATCH 14/14] fix(notifications): scope outcome improvements to backups --- .../tests/notification_recovery_fixture.py | 86 ------- .../tests/test_command_descriptions.py | 4 - .../tests/test_notification_backup_split.py | 116 +++++++++ .../tests/test_notification_corrections.py | 49 +--- .../test_notification_final_corrections.py | 12 +- .../test_notification_maintainer_followup.py | 5 +- .../test_notification_outcome_wording.py | 84 +------ .../test_notification_packaged_recovery.py | 83 ------- .../scripts/tests/test_notification_pve92.py | 4 +- .../test_notification_recovery_corrections.py | 225 ------------------ .../test_notification_recovery_evidence.py | 75 ------ .../tests/test_notification_recovery_order.py | 170 ------------- AppImage/messages/de/common.json | 15 +- AppImage/messages/en/common.json | 2 +- AppImage/messages/es/common.json | 21 +- AppImage/messages/fr/common.json | 15 +- AppImage/messages/it/common.json | 15 +- AppImage/messages/pt/common.json | 15 +- AppImage/messages/sv/common.json | 15 +- AppImage/scripts/build_appimage.sh | 1 - AppImage/scripts/health_monitor.py | 30 +-- AppImage/scripts/health_persistence.py | 90 +------ AppImage/scripts/health_recovery.py | 51 ---- AppImage/scripts/notification_channels.py | 22 +- AppImage/scripts/notification_events.py | 22 +- AppImage/scripts/notification_manager.py | 6 +- AppImage/scripts/notification_templates.py | 51 ++-- .../tests/test_notification_runtime_i18n.py | 17 +- 28 files changed, 231 insertions(+), 1070 deletions(-) delete mode 100644 .github/scripts/tests/notification_recovery_fixture.py create mode 100644 .github/scripts/tests/test_notification_backup_split.py delete mode 100644 .github/scripts/tests/test_notification_packaged_recovery.py delete mode 100644 .github/scripts/tests/test_notification_recovery_corrections.py delete mode 100644 .github/scripts/tests/test_notification_recovery_evidence.py delete mode 100644 .github/scripts/tests/test_notification_recovery_order.py delete mode 100644 AppImage/scripts/health_recovery.py diff --git a/.github/scripts/tests/notification_recovery_fixture.py b/.github/scripts/tests/notification_recovery_fixture.py deleted file mode 100644 index 2cb7fb6f..00000000 --- a/.github/scripts/tests/notification_recovery_fixture.py +++ /dev/null @@ -1,86 +0,0 @@ -"""Pinned, inert recovery review. No operational module imports or threads. -Run with python3 -S under bwrap --unshare-net; DBs and export use TMPDIR. -""" -import ast, contextlib, datetime, hashlib, io, json, os, pathlib, sqlite3, subprocess, sys, tarfile, tempfile, threading, types, typing -from unittest.mock import patch -REPO=pathlib.Path('/home/martino/projects/proxmox/notification-maintainer-followup') -BASE=datetime.datetime(2026,9,30,20,0,0).timestamp() -class Clock(datetime.datetime): - epoch=BASE - tick=0.0 - @classmethod - def now(cls,tz=None): - value=cls.fromtimestamp(cls.epoch,tz) - cls.epoch+=cls.tick - return value -TIME=types.SimpleNamespace(time=lambda:Clock.epoch) - -def extract(path,name,owner,ns): - tree=ast.parse(path.read_text()) - nodes=tree.body if owner is None else next(n.body for n in tree.body if isinstance(n,ast.ClassDef) and n.name==owner) - node=next(n for n in nodes if isinstance(n,(ast.FunctionDef,ast.AsyncFunctionDef)) and n.name==name) - node.decorator_list=[] - exec(compile(ast.Module(body=[node],type_ignores=[]),str(path),'exec'),ns) - return ns[name] - -from notification_fixture import SCRIPTS as scripts -pns=dict(vars(typing),datetime=Clock,timedelta=datetime.timedelta,json=json,sqlite3=sqlite3,contextmanager=contextlib.contextmanager,re=__import__('re'),_re_disk_base=__import__('re')) -pns['disk_base_name']=extract(scripts/'health_persistence.py','disk_base_name',None,pns) -methods={n:extract(scripts/'health_persistence.py',n,'HealthPersistence',pns) for n in ('_get_conn','_db_connection','_init_database','record_error','_record_error_impl','resolve_error','_resolve_error_impl','get_recovery_evidence','_record_event','_entity_from_details','clear_error','get_active_errors','is_error_active','is_error_acknowledged','_get_setting_impl','get_setting','set_setting','acknowledge_error','_acknowledge_error_impl','get_excluded_interface_names')} -def make_store(directory): - s=types.SimpleNamespace(db_path=pathlib.Path(directory)/'health.sqlite',_db_lock=threading.RLock(),DEFAULT_SUPPRESSION_HOURS=24,CATEGORY_SETTING_MAP={}) - for n,f in methods.items(): - if n=='_entity_from_details':setattr(s,n,f) - elif n=='_db_connection':setattr(s,n,types.MethodType(contextlib.contextmanager(f),s)) - else:setattr(s,n,types.MethodType(f,s)) - s._init_database() - return s -def sql(s,query,args=()): - with s._db_connection() as c: - data=c.execute(query,args).fetchall();c.commit();return data -def record(s,key='cpu_usage',category='cpu',reason='CPU high',details=None): - Clock.epoch=BASE-600 - with patch.dict(sys.modules,{'os':types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:True))}):s.record_error(key,category,'WARNING',reason,details) - Clock.epoch=BASE-10 - with patch.dict(sys.modules,{'os':types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:True))}):s.record_error(key,category,'WARNING',reason,details) - Clock.epoch=BASE - assert s.get_active_errors() - return sql(s,'SELECT first_seen FROM errors WHERE error_key=?',(key,))[0][0] -def cpu(s,current=20,history=None,warning=85,critical=95): - ns=dict(vars(typing),time=TIME,os=types.SimpleNamespace(cpu_count=lambda:4),health_persistence=s,psutil=types.SimpleNamespace(cpu_percent=lambda **kw:current,cpu_count=lambda:4)) - fn=extract(scripts/'health_monitor.py','_check_cpu_with_hysteresis','HealthMonitor',ns) - if history is None:history=[{'value':20,'time':BASE-i*10} for i in range(1,11)] - target=types.SimpleNamespace(state_history={'cpu_usage':list(history)},CPU_WARNING=85,CPU_CRITICAL=95,CPU_RECOVERY=75,CPU_WARNING_DURATION=300,CPU_CRITICAL_DURATION=300,CPU_RECOVERY_DURATION=120,_check_cpu_temperature=lambda:None) - refresh=extract(scripts/'health_monitor.py','_refresh_thresholds','HealthMonitor',{}) - with patch.dict(sys.modules,{'health_thresholds':types.SimpleNamespace(get=lambda section,key:({'warning':warning,'critical':critical}.get(key) if section=='cpu' else None))}):refresh(target) - return fn(target) -def poll(s,first,key='cpu_usage',category='cpu',reason='CPU high',details=None,first_done=True,foreign=False,restored=False): - events=[] - meta={'category':category,'reason':reason,'severity':'WARNING','first_seen':first,'details':details} - c=types.SimpleNamespace(_hostname='node-a',_ENTITY_MAP={'cpu':('node',''),'pve_services':('node',''),'network':('node','')},_first_poll_done=first_done,_known_errors={key:meta},_notified_severity={key:'WARNING'},_last_notified={key:BASE-1},SAME_ERROR_COOLDOWN=86400,_get_cooldown_from_db=lambda *a:BASE-1,_queue=types.SimpleNamespace(put=events.append),_guest_storage_error_is_now_foreign=lambda *a:foreign,_save_known_errors_meta=lambda:None) - ns=dict(vars(typing),time=TIME,json=json,re=__import__('re'),NotificationEvent=lambda *a,**kw:types.SimpleNamespace(event_type=a[0],severity=a[1],data=a[2],**kw),startup_grace=types.SimpleNamespace(should_suppress_category=lambda *a:False)) - c._guest_storage_error_is_now_foreign=extract(scripts/'notification_events.py','_guest_storage_error_is_now_foreign','PollingCollector',ns) - fn=extract(scripts/'notification_events.py','_check_persistent_health','PollingCollector',ns) - with patch.dict(sys.modules,{'health_persistence':types.SimpleNamespace(health_persistence=s),'flask_server':types.SimpleNamespace(get_proxmox_node_name=lambda:'node-a',get_cached_pvesh_cluster_resources_vm=lambda:[{'vmid':100,'type':'lxc','node':'node-b' if foreign else 'node-a'}]),'datetime':types.SimpleNamespace(**{**vars(datetime),'datetime':Clock})}): - if restored: - c._KNOWN_ERRORS_SETTING_KEY='pollingcollector_known_errors_v1' - s.set_setting(c._KNOWN_ERRORS_SETTING_KEY,json.dumps(c._known_errors));c._known_errors={} - extract(scripts/'notification_events.py','_load_known_errors_meta','PollingCollector',ns)(c) - c._first_poll_done=bool(c._known_errors) - fn(c) - return [e.data for e in events],c -def service(store, rc=0, stdout='active\n', raised=False, services=('pvedaemon',), clustered=False): - calls=[] - def run(argv, **kw): - calls.append((argv, kw)) - assert argv[:2] == ['systemctl', 'is-active'] - if raised: raise TimeoutError('inert timeout') - return types.SimpleNamespace(returncode=rc, stdout=stdout) - ns=dict(vars(typing),time=TIME,os=types.SimpleNamespace(path=types.SimpleNamespace(exists=lambda p:clustered)),subprocess=types.SimpleNamespace(run=run),health_persistence=store) - result=extract(scripts/'health_monitor.py','_check_pve_services','HealthMonitor',ns)(types.SimpleNamespace(PVE_SERVICES=list(services))) - return result,calls - -@contextlib.contextmanager -def case(): - Clock.epoch=BASE - with tempfile.TemporaryDirectory(prefix='recovery-review-db-',dir=os.environ.get('TMPDIR')) as d:yield make_store(d) diff --git a/.github/scripts/tests/test_command_descriptions.py b/.github/scripts/tests/test_command_descriptions.py index 16e0e9a2..c8fb6431 100644 --- a/.github/scripts/tests/test_command_descriptions.py +++ b/.github/scripts/tests/test_command_descriptions.py @@ -138,14 +138,10 @@ class CommandDescriptionsTests(unittest.TestCase): source = catalog('en')['runtime']['notifications'] for key, value in source['backup'].items(): local.setdefault('backup', {}).setdefault(key, value) - local['channels']['email']['severity'].setdefault( - 'observation', source['channels']['email']['severity']['observation']) local['channels']['email']['status'].setdefault( 'unconfirmed', source['channels']['email']['status']['unconfirmed']) local['channels']['email']['status'].setdefault( 'completed_with_warnings', source['channels']['email']['status']['completed_with_warnings']) - for key, value in source['healthRecovery'].items(): - local.setdefault('healthRecovery', {}).setdefault(key, value) path.write_text(json.dumps(temporary, ensure_ascii=False)) # Model steady state after the bot fills these intentional new # messages; keep repository locales and all other leaves intact. diff --git a/.github/scripts/tests/test_notification_backup_split.py b/.github/scripts/tests/test_notification_backup_split.py new file mode 100644 index 00000000..d63b4e6d --- /dev/null +++ b/.github/scripts/tests/test_notification_backup_split.py @@ -0,0 +1,116 @@ +"""Backup-only maintainer contract at actual render/dispatch/email seams. + +Scope-approved seams: receiver, template lookup, rich enrichment, inert queued +and manual delivery with an email capture sink. No host operations or sends. +""" +import unittest +from notification_fixture import templates, LANGUAGES +from notification_final_fixture import deliver + + +class BackupSplitTests(unittest.TestCase): + def test_recovery_default_is_upstream_resolved_without_proof(self): + for language in LANGUAGES: + data = {'hostname': 'alias', 'category': 'temperature', + 'reason': 'Temperature high (recovered)', 'duration': '2m', + 'original_severity': 'WARNING'} + rendered = templates.render_template('error_resolved', data, language) + expected = templates.runtime_message('templates.error_resolved.title', language, + hostname='alias', category='temperature', entity_suffix='') + self.assertEqual(rendered['title'], expected) + rich, _ = templates.enrich_with_emojis('error_resolved', rendered['title'], rendered['body'], data) + self.assertTrue(rich.startswith('✅ ')) + result = deliver('error_resolved', data, 'OK', language) + self.assertIn('background:#f0fdf4;', result['html']) + + def test_pre_guest_native_subject_keeps_only_cause_once_with_display_alias(self): + from notification_fixture import receive + raw_host = 'pve-production.internal.example' + message = ('Details\n=======\nVMID Name Status Time Size Filename\n' + '\nTotal running time: 0s\nTotal size: 0 B') + for raw in (message, 'ERROR: unable to activate storage PBS\n' + message): + event = receive(raw, 'error', 'vzdump backup status (' + raw_host + + '): backup failed: unable to activate storage PBS') + event.data['hostname'] = 'display-alias {rack.location}' + for language in LANGUAGES: + for manual in (False, True): + result = deliver(event.event_type, event.data, event.severity, language, manual=manual) + self.assertNotIn(raw_host, result['text']) + self.assertNotIn('vzdump backup status', result['text']) + self.assertEqual(result['text'].count('unable to activate storage PBS'), 1) + self.assertIn('display-alias {rack.location}', result['title']) + self.assertEqual(result['data']['pve_title'], event.data['pve_title']) + self.assertEqual(result['data']['pve_message'], raw) + # No global hostname or guest-name substitution: actual diagnostics are + # authoritative, even when they happen to contain the original hostname. + event = receive(message, 'error', 'vzdump backup status (' + raw_host + + '): backup failed: cannot connect to ' + raw_host) + result = deliver(event.event_type, {**event.data, 'hostname': 'alias'}, event.severity) + self.assertEqual(result['body'].count('cannot connect to ' + raw_host), 1) + self.assertNotIn('vzdump backup status', result['body']) + + def test_backup_legacy_keys_keep_meaning_and_unknown_uses_new_keys(self): + # Frozen upstream contract; no Git history required in shipped tests. + legacy = {'en': {'title': '{hostname} → {storage}: Backup complete — {vmname} ({vmid})', 'body': 'Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}', 'label': 'Backup complete'}, 'de': {'title': '{hostname} → {storage}: Sicherung abgeschlossen – {vmname} ({vmid})', 'body': 'Die Sicherung von {vmname} (ID: {vmid}) wurde am {storage} erfolgreich abgeschlossen.\nGröße: {size}', 'label': 'Sicherung abgeschlossen'}, 'es': {'title': '{hostname} → {storage}: Backup completado — {vmname} ({vmid})', 'body': 'El backup de {vmname} (ID: {vmid}) se ha completado correctamente en {storage}.\nTamaño: {size}', 'label': 'Backup completado'}, 'fr': {'title': '{hostname} → {storage}\xa0: Sauvegarde terminée — {vmname} ({vmid})', 'body': "La sauvegarde de {vmname} (ID\xa0: {vmid}) s'est terminée avec succès le {storage}.\nTaille\xa0: {size}", 'label': 'Sauvegarde terminée'}, 'it': {'title': '{hostname} → {storage}: Backup completato — {vmname} ({vmid})', 'body': 'Backup di {vmname} (ID: {vmid}) completato con successo su {storage}.\nTaglia: {size}', 'label': 'Backup completato'}, 'pt': {'title': '{hostname} → {storage}: Backup concluído — {vmname} ({vmid})', 'body': 'Backup de {vmname} (ID: {vmid}) concluído com sucesso em {storage}.\nTamanho: {size}', 'label': 'Backup concluído'}, 'sk': {'title': '{hostname} → {storage}: Záloha dokončená — {vmname} ({vmid})', 'body': 'Záloha {vmname} (ID: {vmid}) na úložisku {storage} bola úspešne dokončená.\nVeľkosť: {size}', 'label': 'Záloha bola dokončená'}, 'sv': {'title': '{hostname} → {storage}: Säkerhetskopiering klar — {vmname} ({vmid})', 'body': 'Säkerhetskopiering av {vmname} (ID: {vmid}) slutfördes framgångsrikt på {storage}.\nStorlek: {size}', 'label': 'Säkerhetskopieringen är klar'}} + for language, expected in legacy.items(): + self.assertEqual(templates._load_runtime_catalog(language)['templates']['backup_complete'], expected) + result = templates.render_template('backup_complete', {'hostname': 'alias {rack.location}'}, language) + self.assertIn('alias {rack.location}', result['title']) + self.assertNotIn('()', result['title']) + new_title = templates.runtime_message('backup.unconfirmedTitle', language, hostname='alias {rack.location}') + self.assertTrue(new_title) + self.assertEqual(result['title'], new_title) + self.assertIn(templates.runtime_message('backup.unconfirmedBody', language), result['body']) + # Absent/blank/non-string translation uses English per-key fallback. + from unittest.mock import patch + english = templates._load_runtime_catalog('en') + for value in (None, '', {}, []): + missing = {'backup': {'unconfirmedTitle': value, 'unconfirmedBody': value}} + with patch.object(templates, '_load_runtime_catalog', side_effect=lambda lang: english if lang == 'en' else missing): + result = templates.render_template('backup_complete', {'hostname': 'alias'}, 'it') + self.assertEqual(result['title'], 'alias: Backup outcome unconfirmed') + self.assertEqual(result['body'], 'The backup outcome is not confirmed.') + # Both catalogs missing: new outcome text still has explicit EN defaults. + with patch.object(templates, '_load_runtime_catalog', return_value={}): + result = templates.render_template('backup_complete', {'hostname': 'alias'}, 'it') + self.assertEqual(result['title'], 'alias: Backup outcome unconfirmed') + self.assertEqual(result['body'], 'The backup outcome is not confirmed.') + + def test_restore_keeps_original_ready_line_and_success_icon_with_warnings(self): + from notification_final_fixture import restore_event + ready = {'en': 'The node is now fully ready to use.', 'de': 'Der Knoten ist nun vollständig einsatzbereit.', 'es': 'El nodo está listo para usarse.', 'fr': 'Le nœud est maintenant entièrement prêt à être utilisé.', 'it': "Il nodo è ora completamente pronto per l'uso.", 'pt': 'O nó agora está totalmente pronto para uso.', 'sk': 'Uzol je teraz úplne pripravený na použitie.', 'sv': 'Noden är nu helt redo att användas.'} + for warning in ('', 'missing module zfs'): + event = restore_event(warning) + for language in LANGUAGES: + result = deliver(event['event_type'], event['data'], event['severity'], language) + self.assertTrue(result['title'].startswith('✅ ')) + self.assertIn(ready[language], result['body']) + self.assertIn(ready[language], result['text']) + self.assertIn('2m', result['text']) + if warning: + self.assertIn(warning, result['text']) + quiet = deliver(event['event_type'], event['data'], event['severity'], language, quiet=True) + self.assertIn('✅', quiet['body']) + self.assertIn(' ' + ready[language], quiet['body']) + self.assertIn('white-space:pre-wrap;', quiet['html']) + + def test_spanish_outcome_and_restore_titles_are_capitalized_and_failure_is_exact(self): + expected = {'confirmed': 'Backup completado', 'completed_with_warnings': 'Backup completado con advertencias', + 'unconfirmed': 'Resultado del backup sin confirmar', 'failed': 'Backup fallido'} + for outcome, title in expected.items(): + result = templates.render_template('backup_complete', {'hostname': 'alias', 'backup_outcome': outcome}, 'es') + self.assertEqual(result['title'], 'alias: ' + title) + failure = templates.render_template('backup_fail', {'hostname': 'alias'}, 'es') + self.assertEqual(failure['title'], 'alias: Backup fallido') + restore = templates.render_template('system_restore_completed', {'hostname': 'alias'}, 'es') + self.assertEqual(restore['title'], 'alias: Restauración del host finalizada') + + def test_recovery_only_quiet_release_keeps_upstream_summary_contract(self): + result = deliver('error_resolved', {'hostname': 'alias', 'category': 'temperature', + 'reason': 'old (recovered)', 'duration': '2m'}, 'OK', quiet=True) + self.assertIn(templates.runtime_message('digest.lead', 'en', count=1).strip(), result['body']) + self.assertIn(templates.runtime_message('digest.footer', 'en'), result['body']) + self.assertIn('✅', result['body']) + + +if __name__ == '__main__': unittest.main() diff --git a/.github/scripts/tests/test_notification_corrections.py b/.github/scripts/tests/test_notification_corrections.py index c9bfaa22..40b2ff58 100644 --- a/.github/scripts/tests/test_notification_corrections.py +++ b/.github/scripts/tests/test_notification_corrections.py @@ -54,8 +54,6 @@ class CorrectionTests(unittest.TestCase): probe.catalogs = {lang: copy.deepcopy(templates._load_runtime_catalog(lang)) for lang in LANGUAGES} parity(probe) # shipped missing keys remain allowed probe.catalogs['sk']['backup'] = copy.deepcopy(probe.catalogs['en']['backup']) - for key in ('observation',): - probe.catalogs['sk']['channels']['email']['severity'][key] = probe.catalogs['en']['channels']['email']['severity'][key] probe.catalogs['sk']['channels']['email']['status']['unconfirmed'] = probe.catalogs['en']['channels']['email']['status']['unconfirmed'] parity(probe) # generation of exactly the pending keys is legal probe.catalogs['sk']['backup']['confirmedTitle'] = 'Missing hostname token' @@ -63,59 +61,14 @@ class CorrectionTests(unittest.TestCase): def test_actual_neutral_style_is_not_success_green(self): - for event, severity, data in (('error_resolved', 'OK', {}), - ('backup_complete', 'INFO', {'backup_outcome': 'unconfirmed'})): + for event, severity, data in (('backup_complete', 'INFO', {'backup_outcome': 'unconfirmed'}),): result, markup = email(event, data, severity) self.assertIn('background:#f9fafb;', markup) self.assertNotIn('background:#f0fdf4;', markup) - def test_actual_manual_caller_carries_event_presentation_context(self): - from notification_fixture import extract, SCRIPTS, EmailChannel - from typing import Optional, Dict, Any - from threading import Lock - captured = [] - channel = object.__new__(EmailChannel) - channel.subject_prefix = '[ProxMenux]' - class Sink: - def send(self, title, body, severity, data): - captured.append((data, channel._format_html(title, body, severity, data))) - return {'success': True} - ns = {'Optional': Optional, 'Dict': Dict, 'Any': Any, 'TEMPLATES': templates.TEMPLATES, - 'resolve_notification_hostname': lambda host, config: host or 'node-a', - 'render_template': templates.render_template, '_should_bypass_ai': lambda event: True} - send = extract(SCRIPTS / 'notification_manager.py', 'send_notification', 'NotificationManager', ns) - class Manager: - _channels = {'email': Sink()} - _config = {} - _lock = Lock() - def _notification_language(self): return 'en' - def is_event_enabled(self, event): return True - def _build_ai_config(self): return {} - def _record_history(self, *args): pass - data = {'category': 'temperature', 'reason': 'old', 'duration': '3d', 'original_severity': 'WARNING', - '_event_type': 'node_reconnect', '_group': 'cluster'} - result = send(Manager(), 'error_resolved', 'OK', '', '', data) - self.assertTrue(result['success']) - context, markup = captured[0] - self.assertEqual(context['_event_type'], 'error_resolved') - self.assertEqual(context['_group'], 'health') - self.assertIn('NO LONGER REPORTED', markup) - self.assertNotIn('>RESOLVED', markup) - self.assertNotIn('color:#16a34a', markup) - self.assertEqual(data['_event_type'], 'node_reconnect') # caller not mutated - def test_disappearance_body_keeps_observation_age_not_green_severity(self): - data = {'hostname': 'node-a', 'category': 'temperature', 'reason': 'old observation', - 'duration': '3d 2h', 'original_severity': 'WARNING', 'severity': 'OK'} - for language in LANGUAGES: - result, markup = email('error_resolved', data, 'OK', language) - for line in result['body'].splitlines(): - if line.strip(): self.assertIn(line.strip(), html.unescape(markup)) - self.assertNotIn('>OK', markup) - self.assertNotIn('color:#16a34a', markup) - self.assertNotIn('>RESOLVED', markup) def test_actual_restore_endpoint_warnings_and_counts_reach_email(self): diff --git a/.github/scripts/tests/test_notification_final_corrections.py b/.github/scripts/tests/test_notification_final_corrections.py index 2976991a..f0bdf551 100644 --- a/.github/scripts/tests/test_notification_final_corrections.py +++ b/.github/scripts/tests/test_notification_final_corrections.py @@ -77,23 +77,13 @@ class FinalCorrectionsTests(unittest.TestCase): self.assertNotIn('script', result['tags']) - def test_long_disappearance_reason_is_present_once_in_actual_dispatch(self): - reason = 'Temperature exceeded configured limit; the source stopped reporting this observation after expiry.' - for language in LANGUAGES: - for manual in (False, True): - result = deliver('error_resolved', {'hostname':'node-a','category':'temperature', - 'reason':reason,'duration':'3d 2h','original_severity':'WARNING'}, 'OK', language, manual=manual) - self.assertEqual(result['text'].count(reason), 1) - self.assertNotIn('>OK', result['html']) - self.assertNotIn('>RESOLVED', result['html']) - def test_raw_restore_and_observation_cells_use_event_scoped_mail_wrapping(self): + def test_raw_restore_cells_use_event_scoped_mail_wrapping(self): token = 'b' * 64 event = restore_event('Boot check: recorded token ' + token + '; verification pending') for language in LANGUAGES: results = [deliver(event['event_type'],event['data'],event['severity'],language,quiet=quiet) for quiet in (False,True)] - results.append(deliver('error_resolved',{'hostname':'node-a','reason':token,'category':'temperature','duration':'3d 2h'},'OK',language)) for result in results: self.assertIn('table-layout:fixed;', result['html']) self.assertIn('word-wrap:break-word;', result['html']) diff --git a/.github/scripts/tests/test_notification_maintainer_followup.py b/.github/scripts/tests/test_notification_maintainer_followup.py index 74bfc2b9..b695659a 100644 --- a/.github/scripts/tests/test_notification_maintainer_followup.py +++ b/.github/scripts/tests/test_notification_maintainer_followup.py @@ -61,7 +61,7 @@ class MaintainerFollowupTests(unittest.TestCase): report+'\nINFO: Starting Backup of VM 100 (lxc)'): self.assertEqual(templates._parse_vzdump_message(message)['vms'][0]['type'],'') - def test_original_subject_only_retained_when_no_guest_context(self): + def test_native_subject_keeps_cause_without_host_envelope(self): subject = 'vzdump backup status (raw-host): backup failed: multiple problems' event = receive('ERROR: archive write failed\n'+NATIVE_REPORT,'error',subject) event.data['hostname']='configured-alias' @@ -72,7 +72,8 @@ class MaintainerFollowupTests(unittest.TestCase): self.assertIn('configured-alias',result['title']) setup=receive('Details\n=======\nVMID Name Status Time Size Filename\n\nTotal running time: 0s\nTotal size: 0 B','error',subject.replace('multiple problems','unable to open storage')) result=deliver(setup.event_type,setup.data,setup.severity) - self.assertEqual(result['text'].count(setup.data['pve_title']),1) + self.assertEqual(result['text'].count('unable to open storage'),1) + self.assertNotIn(setup.data['pve_title'],result['text']) unique=receive(NATIVE_REPORT,'error',subject.replace('multiple problems','job-end hook denied')) result=deliver(unique.event_type,unique.data,unique.severity) self.assertIn('job-end hook denied',result['body']) diff --git a/.github/scripts/tests/test_notification_outcome_wording.py b/.github/scripts/tests/test_notification_outcome_wording.py index 9dabb21f..7d5a35f7 100644 --- a/.github/scripts/tests/test_notification_outcome_wording.py +++ b/.github/scripts/tests/test_notification_outcome_wording.py @@ -15,17 +15,7 @@ from notification_fixture import templates as actual_templates ROOT = Path(__file__).resolve().parents[3] SCRIPTS = ROOT / 'AppImage/scripts' CATALOG = ROOT / 'AppImage/messages/en/common.json' -EXPECTED = { - 'error_resolved': { - 'title': '{hostname}: No longer reported - {category}{entity_suffix}', - 'body': 'The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}', - 'label': 'Recovery notification', - }, - - 'system_restore_completed': { - 'body': 'Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}', - }, -} +EXPECTED = {'system_restore_completed': {'body': 'Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}\nThe node is now fully ready to use.'}} def extract(path, name, owner=None, namespace=None): @@ -82,7 +72,7 @@ class OutcomeWording(unittest.TestCase): result = module.render_template('system_restore_completed', { 'hostname':'node-a','guests':3,'stubs':0,'stale_nodes':0, 'components':1,'duration':'2m','warnings_block':''}, 'es') - self.assertIn('Configuraciones de guests aplicadas: 3', result['body']) + self.assertIn('Guests aplicados: 3', result['body']) self.assertNotIn('invitados', result['body'].lower()) def test_settings_labels_stay_at_upstream_values_in_all_locales(self): @@ -298,11 +288,6 @@ class OutcomeWording(unittest.TestCase): _SEV_DEFAULT = EmailChannel._SEV_DEFAULT subject_prefix = 'ProxMenux' _build_detail_rows = staticmethod(build) - badge = catalog['channels']['email']['severity'].get('observation') or english['channels']['email']['severity']['observation'] - recovery = fmt(Email(), 'No longer reported', 'Body', 'OK', {'_event_type': 'error_resolved', - '_notification_language': lang, '_group': 'health'}) - self.assertIn('>' + badge.upper() + '', recovery) - self.assertIn('color:#6b7280;', recovery) unrelated = fmt(Email(), 'Reconnected', 'Body', 'OK', {'_event_type': 'node_reconnect', '_notification_language': lang, '_group': 'cluster'}) self.assertIn('>' + catalog['channels']['email']['severity']['ok'].upper() + '', unrelated) @@ -346,7 +331,7 @@ class OutcomeWording(unittest.TestCase): self.catalog['runtime']['notifications']['backup'][key]).format(hostname=data['hostname']) self.assertTrue(result['title'].startswith(expected_title), result['title']) else: - self.assertEqual(result['title'], catalog['templates']['backup_complete']['title'].format_map(module._SafeFormatDict(data))) + self.assertEqual(result['title'], module.runtime_message('backup.unconfirmedTitle', lang, hostname=data['hostname'])) self.assertNotIn('{hostname}', result['title']) if state == 'unconfirmed': source = (catalog if catalog.get('backup', {}).get('unconfirmedBody') @@ -357,54 +342,12 @@ class OutcomeWording(unittest.TestCase): self.catalog['runtime']['notifications']['backup']['errorBody'], result['body']) enriched, _ = module.enrich_with_emojis('backup_complete', result['title'], result['body'], data) self.assertTrue(enriched.startswith({'confirmed':'💾✅','unconfirmed':'💾❔','failed':'💾❌'}[state])) - recovery = module.render_template('error_resolved', {'hostname':'node','category':'temperature', - 'reason':'Old observation','duration':'3d','original_severity':'WARNING'}, lang) - recovery_source = catalog - self.assertEqual(recovery['title'], recovery_source['templates']['error_resolved']['title'].format(hostname='node',category='temperature',entity_suffix='')) - self.assertNotIn('resolved', recovery['title'].lower()) if lang == 'en' else None restore = module.render_template('system_restore_completed', {'hostname':'node', 'guests':4, 'stubs':1,'stale_nodes':2,'components':1,'duration':'2m','warnings_block':'Missing module'},lang) self.assertIn('Missing module',restore['body']) - self.assertNotIn('fully ready',restore['body'].lower()) + if lang == 'en': self.assertIn('fully ready',restore['body'].lower()) - def test_stale_record_disappearance_is_not_claimed_recovery(self): - data = {'hostname': 'node-a', 'category': 'temperature', 'reason': 'Temperature observation (no longer reported)', - 'original_severity': 'WARNING', 'duration': '2d 0h', 'severity': 'OK'} - output = self.render('error_resolved', data, 'en') - self.assertIn('no longer in active health records', output['body']) - self.assertNotIn('resolved', (output['title'] + output['body']).lower()) - self.assertIn('Time since first observation', output['body']) - def test_actual_poller_stale_disappearance_keeps_reason_factual(self): - class Store: - def get_active_errors(self): return [] - def is_error_acknowledged(self, key): return False - class Event: - def __init__(self, *args, **kwargs): self.kind, self.severity, self.data = args[:3] - class Queue: - def __init__(self): self.items = [] - def put(self, event): self.items.append(event) - ns = {'time': time, 'json': json, 'NotificationEvent': Event, 'Dict': dict} - poll = extract(SCRIPTS / 'notification_events.py', '_check_persistent_health', 'PollingCollector', ns) - class Collector: - _hostname = 'node-a' - _ENTITY_MAP = {'temperature': ('node', '')} - _first_poll_done = True - _known_errors = {'temp': {'category': 'temperature', 'reason': 'Temperature high', - 'severity': 'WARNING', 'first_seen': '2026-09-25T00:00:00'}} - _notified_severity = {'temp': 'WARNING'} - _last_notified = {'temp': 1} - _queue = Queue() - def _guest_storage_error_is_now_foreign(self, *a): return False - def _save_known_errors_meta(self): pass - with patch.dict(sys.modules, {'health_persistence': types.SimpleNamespace(health_persistence=Store())}): - poll(Collector()) - events = Collector._queue.items - self.assertEqual(len(events), 1) - self.assertEqual((events[0].kind, events[0].severity), ('error_resolved', 'OK')) - self.assertEqual(events[0].data['reason'], 'Temperature high (no longer reported)') - rendered = self.render(events[0].kind, events[0].data, 'en') - self.assertNotIn('recovered', rendered['body'].lower()) def test_warning_and_clean_restore_keep_only_reported_outcome(self): for warnings in ('', '⚠️ Boot sanity: missing modules\n'): @@ -423,24 +366,9 @@ class OutcomeWording(unittest.TestCase): self.assertEqual(event['severity'], 'WARNING' if warnings else 'INFO') result = self.render(event['event_type'], event['data'], 'en') self.assertIn('Post-restore tasks completed', result['body']) - self.assertNotIn('fully ready', result['body']) + self.assertIn('fully ready', result['body']) if warnings: self.assertIn('missing modules', result['body']) - def test_missing_key_fallback_and_synthetic_translation(self): - catalog = copy.deepcopy(self.catalog) - translated = copy.deepcopy(self.catalog) - for event, fields in EXPECTED.items(): - for field in fields: translated['runtime']['notifications']['templates'][event].pop(field) - _, render = renderer(catalog, translated) - for event, fields in EXPECTED.items(): - for field in fields: - if field not in ('title', 'body'): - continue - self.assertEqual(render(event, {'category': 'disk'}, 'it')[field], - render(event, {'category': 'disk'}, 'en')[field]) - translated['runtime']['notifications']['templates']['error_resolved']['title'] = 'Synthetic observation: {category}' - _, render = renderer(catalog, translated) - self.assertEqual(render('error_resolved', {'category': 'disk'}, 'it')['title'], 'Synthetic observation: disk') def test_rich_backup_icon_tracks_outcome_and_digest_default_is_neutral(self): tree = ast.parse((SCRIPTS / 'notification_templates.py').read_text()) @@ -465,8 +393,6 @@ class OutcomeWording(unittest.TestCase): tree = ast.parse((SCRIPTS / 'notification_templates.py').read_text()) icon_map = ast.literal_eval(next(n.value for n in tree.body if isinstance(n, ast.Assign) and any(isinstance(t, ast.Name) and t.id == 'EVENT_EMOJI' for t in n.targets))) - for event in ('error_resolved', 'system_restore_completed'): - self.assertNotIn('✅', icon_map[event], event) self.assertNotIn('✅', icon_map['backup_complete']) # buffered digest has no outcome metadata diff --git a/.github/scripts/tests/test_notification_packaged_recovery.py b/.github/scripts/tests/test_notification_packaged_recovery.py deleted file mode 100644 index 0c359634..00000000 --- a/.github/scripts/tests/test_notification_packaged_recovery.py +++ /dev/null @@ -1,83 +0,0 @@ -"""Execute the build's literal backend copies, then a clean shipped-only runtime.""" -import os -from pathlib import Path -import re -import subprocess -import sys -import tempfile -import unittest - -ROOT = Path(__file__).resolve().parents[3] -PROBE = r''' -import ast, pathlib, sys, types, typing -stage = pathlib.Path(sys.argv[1]) -sys.path.insert(0, str(stage)) -assert 'health_recovery' not in sys.modules -import notification_templates as templates -from notification_channels import EmailChannel -import health_recovery -for module in (templates, health_recovery, sys.modules['notification_channels']): - assert pathlib.Path(module.__file__).parent == stage -assert not any('projects/proxmox' in p for p in sys.path) -templates._get_hostname = lambda: 'node-a' -channel = object.__new__(EmailChannel) -channel.subject_prefix = '[ProxMenux]' -neutral = {'hostname': 'node-a', 'category': 'cpu', 'reason': 'CPU high', '_event_type': 'error_resolved'} -templates.render_template('error_resolved', neutral) -templates.enrich_with_emojis('error_resolved', 'Observation', 'Body', neutral) -channel._format_html('Observation', 'Body', 'OK', neutral) -templates.render_template('node_reconnect', {'hostname': 'node-a'}) -now = 1000.0 -calls = [] -ns = dict(vars(typing), time=types.SimpleNamespace(time=lambda: now), - os=types.SimpleNamespace(cpu_count=lambda: 4), - psutil=types.SimpleNamespace(cpu_percent=lambda **kw: 20, cpu_count=lambda: 4), - health_persistence=types.SimpleNamespace(resolve_error=lambda *a, **kw: calls.append((a, kw)))) -tree = ast.parse((stage/'health_monitor.py').read_text()) -owner = next(n for n in tree.body if isinstance(n, ast.ClassDef) and n.name == 'HealthMonitor') -node = next(n for n in owner.body if isinstance(n, ast.FunctionDef) and n.name == '_check_cpu_with_hysteresis') -exec(compile(ast.Module(body=[node], type_ignores=[]), 'shipped_cpu', 'exec'), ns) -monitor = types.SimpleNamespace(state_history={'cpu_usage': [{'value':20,'time':now-i*10} for i in range(1,11)]}, - CPU_WARNING=85, CPU_CRITICAL=95, CPU_RECOVERY=75, CPU_WARNING_DURATION=300, - CPU_CRITICAL_DURATION=300, CPU_RECOVERY_DURATION=120, _check_cpu_temperature=lambda:None) -assert ns['_check_cpu_with_hysteresis'](monitor)['status'] == 'OK' -assert len(calls) == 1 and calls[0][1]['check_evidence']['value'] == 20 -proof = calls[0][1]['check_evidence'] -native = dict(neutral, error_key='cpu_usage', check_evidence=proof, is_recovery=True, recovery_outcome='resolved') -# Actual body/icon/email consumers from the package, no source-module fixture rescue. -import unittest.mock -with unittest.mock.patch('health_recovery.time.time', return_value=now): - result = templates.render_template('error_resolved', native) - title, body = result['title'], result['body'] - assert 'Resolved' in title and 'fresh health check' in body, (title, body, proof) - rich_title, rich_body = templates.enrich_with_emojis('error_resolved', title, body, native) - assert rich_title.startswith('✅') - assert 'background:#f0fdf4;' in channel._format_html(title, body, 'OK', native) -print('shipped-only neutral body/icon/email + CPU native measurement/proof consumers PASS') -''' - - -class PackagedRecoveryTests(unittest.TestCase): - def test_actual_copy_manifest_supports_isolated_recovery_runtime(self): - source = ROOT / 'AppImage/scripts' - build = (source / 'build_appimage.sh').read_text() - lines = [line for line in build.splitlines() - if re.match(r'^cp "\$SCRIPT_DIR/[^"/]+\.py" "\$APP_DIR/usr/bin/"', line)] - self.assertTrue(lines) - catalog_copy = re.search(r'^for locale in en de es fr it pt sk sv; do\n.*?^done$', build, re.MULTILINE | re.DOTALL) - if catalog_copy is None: - self.fail('The shipped locale-copy loop was not found in the actual build script') - lines.append(catalog_copy.group(0)) - with tempfile.TemporaryDirectory(prefix='shipped-recovery-') as directory: - stage = Path(directory) / 'usr/bin' - stage.mkdir(parents=True) - copied = subprocess.run(['/bin/bash'], input='set -e\n'+'\n'.join(lines)+'\n', text=True, - capture_output=True, env={**os.environ, 'SCRIPT_DIR': str(source), 'APPIMAGE_ROOT': str(source.parent), 'APP_DIR': directory}) - self.assertEqual(copied.returncode, 0, copied.stderr) - result = subprocess.run([sys.executable, '-I', '-B', '-c', PROBE, str(stage)], - cwd=directory, text=True, capture_output=True) - self.assertEqual(result.returncode, 0, result.stdout + result.stderr) - - -if __name__ == '__main__': - unittest.main() diff --git a/.github/scripts/tests/test_notification_pve92.py b/.github/scripts/tests/test_notification_pve92.py index 220d129a..d0f781a4 100644 --- a/.github/scripts/tests/test_notification_pve92.py +++ b/.github/scripts/tests/test_notification_pve92.py @@ -103,9 +103,7 @@ class PVE92Tests(unittest.TestCase): 'warnings_block': ''} slovak = templates._load_runtime_catalog('sk') english = templates._load_runtime_catalog('en') - for event, field in (('error_resolved', 'title'), ('error_resolved', 'body'), - ('system_restore_completed', 'body'), - ('backup_complete', 'title'), ('backup_complete', 'body')): + for event, field in (('system_restore_completed', 'body'),): with self.subTest(event=event, field=field): value = slovak['templates'][event][field] result = templates.render_template(event, data, 'sk') diff --git a/.github/scripts/tests/test_notification_recovery_corrections.py b/.github/scripts/tests/test_notification_recovery_corrections.py deleted file mode 100644 index 6908cbfb..00000000 --- a/.github/scripts/tests/test_notification_recovery_corrections.py +++ /dev/null @@ -1,225 +0,0 @@ -"""Native initializer, measurement methods, SQL writers/readers and collector.""" -import unittest -from notification_recovery_fixture import case, Clock, BASE, cpu, sql, poll - -class RecoveryCorrectionTests(unittest.TestCase): - def test_supported_low_warning_current_violation_is_neutral(self): - with case() as store: - Clock.epoch = BASE - 600 - initial = cpu(store, 80, [{'value':80, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=50) - self.assertEqual(initial['status'], 'WARNING') - first = sql(store, 'SELECT first_seen FROM errors')[0][0] - Clock.epoch = BASE - result = cpu(store, 60, [{'value':60, 'time':BASE-i*5} for i in range(1,26)], warning=50) - events, _ = poll(store, first, reason=initial['reason']) - # Preserve operational clear/hysteresis behavior, not its factual claim. - self.assertEqual(result['status'], 'OK') - self.assertFalse(events[0]['is_recovery'], events) - self.assertIsNone(store.get_recovery_evidence('cpu_usage', first)) - - def test_original_policy_survives_repeated_native_updates(self): - import json - for later_warning in (85, 40): - with self.subTest(later_warning=later_warning), case() as store: - Clock.epoch = BASE - 600 - cpu(store, 80, [{'value':80, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=50) - first = sql(store, 'SELECT first_seen FROM errors')[0][0] - for step in (400, 200): - Clock.epoch = BASE-step - cpu(store, 90, [{'value':90, 'time':Clock.epoch-i*5} for i in range(1,26)], warning=later_warning) - original = json.loads(sql(store, 'SELECT details FROM errors')[0][0])['cpu_policy'] - self.assertEqual(original['warning'], 50) - Clock.epoch = BASE - cpu(store, 20, warning=later_warning) - self.assertIsNone(store.get_recovery_evidence('cpu_usage', first)) - self.assertFalse(poll(store, first)[0][0]['is_recovery']) - - def test_clock_rollback_new_row_generic_clear_cannot_inherit_old_proof(self): - for rollback, reuse_first in ((True,False),(False,False),(True,True)): - with self.subTest(rollback=rollback,reuse_first=reuse_first),case() as store: - Clock.epoch = BASE-600 - cpu(store, 90, [{'value':90, 'time':Clock.epoch-i*5} for i in range(1,26)]) - first = sql(store, 'SELECT first_seen FROM errors')[0][0] - Clock.epoch = BASE; Clock.tick = .001 - try: cpu(store) - finally: Clock.tick = 0 - self.assertTrue(store.get_recovery_evidence('cpu_usage', first)) - store.acknowledge_error('cpu_usage', suppression_hours=-1) - store.clear_error('cpu_usage') - Clock.epoch = BASE-100 if rollback else BASE+1 - store.record_error('cpu_usage','cpu','WARNING','new incident after clock step',{}) - second = sql(store, 'SELECT first_seen FROM errors')[0][0] - self.assertNotEqual(first, second) - if reuse_first: - # Restored malformed snapshot with reused wall-clock identity; - # native row id and latest closure still prevent replay. - sql(store,'UPDATE errors SET first_seen=?',(first,));second=first - Clock.epoch = BASE+.0005 if rollback else BASE+2 - store.clear_error('cpu_usage'); Clock.epoch = BASE+3 - self.assertIsNone(store.get_recovery_evidence('cpu_usage', second)) - self.assertFalse(poll(store, second)[0][0]['is_recovery']) - - def test_malformed_native_and_manual_proof_is_neutral_at_all_consumers(self): - import copy, json, time - from unittest.mock import patch - from notification_fixture import LANGUAGES - from notification_final_fixture import deliver - with case() as store: - Clock.epoch = BASE-600 - cpu(store, 90, [{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)]) - first = sql(store,'SELECT first_seen FROM errors')[0][0] - Clock.epoch = BASE; cpu(store) - event_data = json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0]) - data = poll(store,first)[0][0] - for field, bad in [('checked_at','bad'),('checked_at',True),('checked_at',float('inf')), - ('checked_at',10**400),('value',60),('max_sample',float('nan')), - ('normal_samples',True),('normal_samples',9),('checked_at',BASE+1),('checked_at',BASE-7201), - ('policy',{'warning':False,'critical':95,'recovery':75}), - ('policy',{'warning':96,'critical':95,'recovery':75}), - ('policy',{'warning':85,'critical':95}), - ('policy',{'warning':85,'critical':95,'recovery':float('nan')}), - ('policy',{'warning':85,'critical':95,'recovery':75,'extra':0})]: - broken = copy.deepcopy(event_data) - broken['check_evidence'][field] = bad - if field == 'value': broken['check_evidence']['policy']['warning'] = 50 - sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps(broken),)) - with self.subTest(field=field,bad=str(bad)),patch('health_recovery.time.time',return_value=BASE): - self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) - for language in LANGUAGES: - for manual in (False,True): - result = deliver('error_resolved',{**data,'check_evidence':broken['check_evidence']},'OK',language,manual=manual) - self.assertNotIn('background:#f0fdf4;', result['html']) - # Caller content is trusted, not authenticated native proof; even - # well-shaped assertions require a valid time/type/numeric contract. - for proof in ({'check':'cpu_usage'}, {'check':'cpu_usage','checked_at':time.time()+1}): - self.assertNotIn('background:#f0fdf4;', deliver('error_resolved',{**data,'check_evidence':proof},'OK',manual=True)['html']) - - def test_exact_service_active_native_clear_reaches_recovery_consumers(self): - from unittest.mock import patch - from notification_recovery_fixture import service - from notification_fixture import LANGUAGES - from notification_final_fixture import deliver - with case() as store: - Clock.epoch = BASE-600 - service(store,3,'inactive\n') - first = sql(store,'SELECT first_seen FROM errors')[0][0] - Clock.epoch = BASE - result, calls = service(store) - self.assertEqual(result['status'],'OK') - self.assertEqual(calls,[(['systemctl','is-active','pvedaemon'], {'capture_output':True,'text':True,'timeout':2})]) - proof = store.get_recovery_evidence('pve_service_pvedaemon',first) - self.assertTrue(proof) - data = poll(store,first,'pve_service_pvedaemon','pve_services','PVE service pvedaemon is inactive',details={'service':'pvedaemon'})[0][0] - self.assertTrue(data['is_recovery']) - with patch('health_recovery.time.time',return_value=BASE): - for language in LANGUAGES: - for manual in (False,True): - rendered = deliver('error_resolved',data,'OK',language,manual=manual) - self.assertIn('background:#f0fdf4;',rendered['html']) - self.assertIn('pvedaemon', rendered['text']) - - def test_cpu_positive_default_low_policy_and_neutral_history_controls(self): - from notification_recovery_fixture import record - for warning,current,history,legacy,expected in ( - (85,20,None,False,True), (50,20,None,False,True), - (85,20,None,True,False), (85,99,[],False,False), - (85,20,[],False,False), - (85,20,[{'value':20,'time':BASE+i*5} for i in range(1,10)],False,False), - (85,20,[{'value':20,'time':BASE-121-i} for i in range(12)],False,False), - (85,99,None,False,False), - (85,float('nan'),None,False,False), (85,float('inf'),None,False,False), - (85,True,None,False,False), (85,10**400,None,False,False)): - with self.subTest(warning=warning,current=str(current),legacy=legacy), case() as store: - if legacy: first = record(store) - else: - Clock.epoch=BASE-600 - cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)],warning=warning) - first=sql(store,'SELECT first_seen FROM errors')[0][0] - Clock.epoch=BASE; cpu(store,current,history,warning=warning) - self.assertEqual(bool(store.get_recovery_evidence('cpu_usage',first)),expected) - - def test_service_unavailable_removed_overall_ok_and_ack_controls(self): - from notification_recovery_fixture import service, record - for rc,stdout,raised in ((3,'inactive\n',False),(4,'unknown\n',False),(0,'active extra\n',False),(1,'active\n',False),(0,'',True)): - with self.subTest(rc=rc,stdout=stdout,raised=raised),case() as store: - Clock.epoch=BASE-600; service(store,3,'inactive\n') - first=sql(store,'SELECT first_seen FROM errors')[0][0] - Clock.epoch=BASE; service(store,rc,stdout,raised) - self.assertIsNone(store.get_recovery_evidence('pve_service_pvedaemon',first)) - self.assertFalse(poll(store,first,'pve_service_pvedaemon','pve_services')[0]) - for removed in ('pvedaemon','corosync'): - with self.subTest(removed=removed),case() as store: - first=record(store,'pve_service_'+removed,'pve_services','service inactive',{'service':removed}) - result,calls=service(store,services=(),clustered=False) - self.assertEqual(result['status'],'OK'); self.assertEqual(calls,[]) - store.clear_error('pve_service_'+removed) - self.assertFalse(poll(store,first,'pve_service_'+removed,'pve_services')[0][0]['is_recovery']) - with case() as store: - first=record(store,'pve_service_corosync','pve_services','corosync inactive',{'service':'corosync'}) - result,calls=service(store,services=('pvedaemon',),clustered=False) - self.assertEqual(result['status'],'OK') - self.assertEqual([c[0][-1] for c in calls],['pvedaemon']) - self.assertTrue(store.is_error_active('pve_service_corosync')) - self.assertIsNone(store.get_recovery_evidence('pve_service_corosync',first)) - with case() as store: - first=record(store,'pve_service_pvedaemon','pve_services','service inactive') - store.acknowledge_error('pve_service_pvedaemon',suppression_hours=-1) - store.clear_error('pve_service_pvedaemon',check_evidence={'check':'pve_service_pvedaemon','checked_at':BASE,'service':'pvedaemon','state':'active','returncode':0}) - self.assertEqual(sql(store,'SELECT id FROM errors'),[]) - self.assertIsNone(store.get_recovery_evidence('pve_service_pvedaemon',first)) - - def test_native_binding_latest_closure_rollbacks_and_consistent_ack_read(self): - import contextlib, json, sqlite3 - for mutation in ('row_id','first_seen','closure','last_seen','latest_clear','latest_resolve','ack','event_insert_failure','ack_before_join'): - with self.subTest(mutation=mutation),case() as store: - Clock.epoch=BASE-600 - cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)]) - first=sql(store,'SELECT first_seen FROM errors')[0][0] - Clock.epoch=BASE - if mutation=='event_insert_failure': - sql(store,"CREATE TRIGGER fail_resolve BEFORE INSERT ON events WHEN NEW.event_type='resolved' BEGIN SELECT RAISE(ABORT,'fixture'); END") - self.assertEqual(cpu(store)['status'],'UNKNOWN') - self.assertIsNone(sql(store,'SELECT resolved_at FROM errors')[0][0]) - continue - cpu(store); self.assertTrue(store.get_recovery_evidence('cpu_usage',first)) - if mutation in ('row_id','first_seen','closure'): - data=json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0]) - field={'row_id':'id','first_seen':'first_seen','closure':'resolved_at'}[mutation] - data['incident'][field]='wrong' - sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps(data),)) - elif mutation=='last_seen': sql(store,'UPDATE errors SET last_seen=?',(Clock.fromtimestamp(BASE+1).isoformat(),)) - elif mutation.startswith('latest_'): - sql(store,"INSERT INTO events(event_type,error_key,timestamp,data) VALUES(?,'cpu_usage',?,'{}')",('cleared' if mutation=='latest_clear' else 'resolved',Clock.now().isoformat())) - elif mutation=='ack': store.acknowledge_error('cpu_usage',suppression_hours=-1) - else: - original=store._db_connection; triggered=[] - class Proxy: - def __init__(self,connection):self.connection=connection - def __getattr__(self,name):return getattr(self.connection,name) - def execute(self,query,args=()): - if 'FROM errors e JOIN events' in query and not triggered: - triggered.append(True) - store.acknowledge_error('cpu_usage',suppression_hours=-1) - return self.connection.execute(query,args) - @contextlib.contextmanager - def interleaved(**kw): - with original(**kw) as conn: yield Proxy(conn) - store._db_connection=interleaved - self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) - if mutation=='ack_before_join': self.assertTrue(triggered) - with case() as store: - sql(store,"INSERT INTO errors(error_key,category,severity,reason,first_seen,last_seen) VALUES('cpu_usage','cpu','WARNING','fixture','x','x')") - with self.assertRaises(sqlite3.IntegrityError): - sql(store,"INSERT INTO errors(error_key,category,severity,reason,first_seen,last_seen) VALUES('cpu_usage','cpu','WARNING','duplicate','x','x')") - - def test_malformed_history_declines_proof_without_changing_operational_clear(self): - with case() as store: - Clock.epoch=BASE-600 - cpu(store,90,[{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)]) - first=sql(store,'SELECT first_seen FROM errors')[0][0] - Clock.epoch=BASE - result=cpu(store,20,[{'value':10**400,'time':BASE-i*5} for i in range(1,10)]) - self.assertEqual(result['status'],'OK') - self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) - -if __name__ == '__main__': unittest.main() diff --git a/.github/scripts/tests/test_notification_recovery_evidence.py b/.github/scripts/tests/test_notification_recovery_evidence.py deleted file mode 100644 index 096feed2..00000000 --- a/.github/scripts/tests/test_notification_recovery_evidence.py +++ /dev/null @@ -1,75 +0,0 @@ -"""Fresh existing-check provenance; native initializer and disposable SQLite.""" -import json -import time -import unittest -from unittest.mock import patch -from notification_fixture import templates, LANGUAGES -from notification_final_fixture import deliver -from notification_recovery_fixture import case, Clock, BASE, cpu, sql, poll - - -def original_cpu(store): - Clock.epoch = BASE-600 - result = cpu(store, 90, [{'value':90,'time':Clock.epoch-i*5} for i in range(1,26)]) - assert result['status'] == 'WARNING' - Clock.epoch = BASE - return sql(store, 'SELECT first_seen FROM errors')[0][0] - - -class RecoveryEvidenceTests(unittest.TestCase): - def test_cpu_success_provenance_is_persisted_only_after_normal_samples(self): - with case() as store: - first = original_cpu(store) - self.assertEqual(cpu(store)['status'], 'OK') - proof = store.get_recovery_evidence('cpu_usage', first) - self.assertTrue(proof) - self.assertEqual(proof['check'], 'cpu_usage') - self.assertEqual(proof['checked_at'], BASE) - # A generic closure never gains proof; use another actual native row. - store.record_error('pve_service_test','pve_services','CRITICAL','inactive') - store.resolve_error('pve_service_test','No longer present') - self.assertFalse(json.loads(sql(store,"SELECT data FROM events ORDER BY id DESC LIMIT 1")[0][0]).get('check_evidence')) - - def test_recovery_query_requires_fresh_same_incident_proof(self): - with case() as store: - first = original_cpu(store); cpu(store) - proof = store.get_recovery_evidence('cpu_usage',first) - self.assertTrue(proof) - self.assertIsNone(store.get_recovery_evidence('cpu_usage','different incident')) - saved = sql(store,'SELECT last_seen,resolved_at FROM errors')[0] - for field,value in [('acknowledged',1),('resolved_at',None),('last_seen',Clock.fromtimestamp(BASE+1).isoformat())]: - sql(store,f'UPDATE errors SET {field}=?',(value,)) - self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) - sql(store,'UPDATE errors SET acknowledged=0,last_seen=?,resolved_at=?',saved) - data = json.loads(sql(store,"SELECT data FROM events WHERE event_type='resolved'")[0][0]) - for bad in (None,{'check':'cpu_usage','checked_at':BASE-7201}, {'check':'storage_removed','checked_at':BASE}, {'check':'cpu_usage','checked_at':float('inf')}, {'check':'cpu_usage','checked_at':10**400}): - sql(store,"UPDATE events SET data=? WHERE event_type='resolved'",(json.dumps({**data,'check_evidence':bad}),)) - self.assertIsNone(store.get_recovery_evidence('cpu_usage',first)) - - def test_poller_and_all_consumers_distinguish_proven_recovery_from_disappearance(self): - for proved in (False,True): - with case() as store: - first = original_cpu(store) - if proved: cpu(store) - else: store.resolve_error('cpu_usage','No longer present') - data = poll(store,first,reason='CPU high')[0][0] - self.assertEqual(data['is_recovery'],proved) - self.assertEqual(data['recovery_outcome'],'resolved' if proved else 'no_longer_reported') - with patch('health_recovery.time.time',return_value=BASE): - for lang in LANGUAGES: - for manual in (False,True): - result = deliver('error_resolved',data,'OK',lang,manual=manual) - self.assertEqual('background:#f0fdf4;' in result['html'],proved) - if proved: - self.assertIn(templates.runtime_message('healthRecovery.title',lang,hostname='node-a',category='cpu',entity_suffix=''),result['title']) - self.assertEqual(result['text'].count(data['reason']),1) - - def test_manual_recovery_flag_alone_is_not_authoritative_evidence(self): - data={'hostname':'node-a','category':'cpu','reason':'Observation disappeared','duration':'1h','original_severity':'WARNING','recovery_outcome':'resolved'} - for lang in LANGUAGES: - for manual in (False,True): - result=deliver('error_resolved',data,'OK',lang,manual=manual) - self.assertNotIn('background:#f0fdf4;',result['html']) - self.assertNotIn(templates.runtime_message('healthRecovery.body',lang,**data),result['body']) - -if __name__=='__main__':unittest.main() diff --git a/.github/scripts/tests/test_notification_recovery_order.py b/.github/scripts/tests/test_notification_recovery_order.py deleted file mode 100644 index 9b3e1607..00000000 --- a/.github/scripts/tests/test_notification_recovery_order.py +++ /dev/null @@ -1,170 +0,0 @@ -"""Durable native observation order, not wall-clock row identity, admits proof.""" -import datetime -import json -import re -import sys -import types -import typing -import unittest -from unittest.mock import patch -from notification_recovery_fixture import case, Clock, BASE, cpu, service, sql, extract, scripts, TIME -from notification_fixture import LANGUAGES -from notification_final_fixture import deliver - - -def collector(store): - events = [] - target = types.SimpleNamespace(_hostname='alias {rack.location}', - _ENTITY_MAP={'cpu':('node',''), 'pve_services':('node','')}, _first_poll_done=False, - _known_errors={}, _notified_severity={}, _last_notified={}, SAME_ERROR_COOLDOWN=86400, - _get_cooldown_from_db=lambda *a:BASE-1, _queue=types.SimpleNamespace(put=events.append), - _save_known_errors_meta=lambda:None) - ns = dict(vars(typing), time=TIME, json=json, re=re, - NotificationEvent=lambda *a, **kw:types.SimpleNamespace(event_type=a[0], severity=a[1], data=a[2], **kw), - startup_grace=types.SimpleNamespace(should_suppress_category=lambda *a:False)) - target._guest_storage_error_is_now_foreign = extract(scripts/'notification_events.py', - '_guest_storage_error_is_now_foreign', 'PollingCollector', ns) - poll = extract(scripts/'notification_events.py', '_check_persistent_health', 'PollingCollector', ns) - def tick(): - with patch.dict(sys.modules, {'health_persistence':types.SimpleNamespace(health_persistence=store), - 'datetime':types.SimpleNamespace(**{**vars(datetime), 'datetime':Clock})}): - poll(target) - return target, events, tick - - -def abnormal(store, kind, value=99): - if kind == 'cpu': - return cpu(store, value, [{'value':value, 'time':Clock.epoch-i*5} for i in range(1,26)]) - return service(store,3,'inactive\n')[0] - - -def normal(store, kind): - return cpu(store) if kind == 'cpu' else service(store)[0] - - -def initial_closure(store, kind): - key = 'cpu_usage' if kind == 'cpu' else 'pve_service_pvedaemon' - Clock.epoch = BASE-600 - abnormal(store,kind,90) - target, events, tick = collector(store) - tick() - assert target._first_poll_done and key in target._known_errors and not events - snapshot = json.loads(json.dumps(target._known_errors)) - Clock.epoch = BASE - assert normal(store,kind)['status'] == 'OK' - first = snapshot[key]['first_seen'] - assert store.get_recovery_evidence(key,first) - return key, first, snapshot, target, events, tick - - -class RecoveryOrderTests(unittest.TestCase): - def assert_consumers(self, data, expected): - with patch('health_recovery.time.time', return_value=Clock.epoch): - for language in LANGUAGES: - for manual in (False,True): - with self.subTest(language=language, manual=manual): - result = deliver('error_resolved',data,'OK',language,manual=manual) - self.assertEqual('background:#f0fdf4;' in result['html'],expected) - self.assertIn('alias {rack.location}',result['text']) - quiet = deliver('error_resolved',data,'OK',language,quiet=True) - self.assertEqual(len(quiet['buffered']),1) - # The existing quiet digest stores the rendered title, not - # health body/proof metadata. Assert its exact outcome label. - from notification_fixture import templates - label = templates.render_template('error_resolved',data,language)['title'].split(': ',1)[-1] - self.assertIn(label, quiet['body']) - self.assertIn(label, quiet['buffered'][0][2]) - - def assert_superseded(self, kind): - for offset in (-100,0,1): - for value in ((90,99) if kind == 'cpu' else (99,)): - with self.subTest(kind=kind, offset=offset, value=value), case() as store: - key, first, snapshot, target, events, tick = initial_closure(store,kind) - oldrow = sql(store,'SELECT id,first_seen,resolved_at FROM errors')[0] - prior = sql(store,'SELECT id,event_type FROM events ORDER BY id') - Clock.epoch = BASE+offset - renewed = abnormal(store,kind,value) - self.assertEqual(renewed['status'],'WARNING' if kind == 'cpu' and value == 90 else 'CRITICAL') - self.assertEqual(sql(store,'SELECT id,first_seen,resolved_at FROM errors')[0],oldrow) - later = sql(store,'SELECT id,event_type FROM events ORDER BY id')[-1] - self.assertGreater(later[0],prior[-1][0]) - self.assertEqual(later[1],'escalated' if kind == 'cpu' and value == 99 else 'updated') - if kind == 'cpu': - policy=json.loads(sql(store,'SELECT details FROM errors')[0][0])['cpu_policy'] - self.assertEqual(policy,{'warning':85,'critical':95,'recovery':75}) - Clock.epoch = BASE+2 - tick() - self.assertEqual(len(events),1) - data = events[0].data - self.assertFalse(data['is_recovery']) - self.assertIsNone(store.get_recovery_evidence(key,first)) - self.assert_consumers(data,False) - # Existing operations do not rearm the resolved row, so a - # later normal check cannot establish a NEW native closure. - before = sql(store,'SELECT id FROM events ORDER BY id') - Clock.epoch = BASE+3 - self.assertEqual(normal(store,kind)['status'],'OK') - self.assertEqual(sql(store,'SELECT id FROM events ORDER BY id'),before) - self.assertIsNone(store.get_recovery_evidence(key,first)) - - def test_cpu_superseded_same_row_is_neutral_at_all_consumers(self): - self.assert_superseded('cpu') - - def test_service_superseded_same_row_is_neutral_at_all_consumers(self): - self.assert_superseded('service') - - def test_fresh_closure_and_repeated_noop_clear_keep_genuine_proof(self): - for kind in ('cpu','service'): - with self.subTest(kind=kind),case() as store: - key, first, snapshot, target, events, tick = initial_closure(store,kind) - before = sql(store,'SELECT id,event_type FROM events ORDER BY id') - Clock.epoch = BASE+1 - for _ in range(3): - store.clear_error(key) - store.resolve_error(key,'generic repeat') - normal(store,kind) - self.assertEqual(sql(store,'SELECT id,event_type FROM events ORDER BY id'),before) - self.assertTrue(store.get_recovery_evidence(key,first)) - tick() - self.assertEqual(len(events),1) - self.assertTrue(events[0].data['is_recovery']) - self.assert_consumers(events[0].data,True) - tick() - self.assertEqual(len(events),1) - - def test_later_actual_generic_closure_blocks_older_proof_even_clock_rollback(self): - for kind in ('cpu','service'): - with self.subTest(kind=kind),case() as store: - key, first, *_ = initial_closure(store,kind) - Clock.epoch = BASE-100 - with store._db_connection() as conn: - store._record_event(conn.cursor(),'cleared',key,{'reason':'generic actual closure','check_evidence':None}) - conn.commit() - self.assertIsNone(store.get_recovery_evidence(key,first)) - - def test_explicit_acknowledged_row_suppresses_native_proof(self): - for kind in ('cpu','service'): - with self.subTest(kind=kind),case() as store: - key, first, snapshot, target, events, tick = initial_closure(store,kind) - store.acknowledge_error(key,suppression_hours=-1) - self.assertEqual(sql(store,'SELECT acknowledged FROM errors')[0][0],1) - self.assertIsNone(store.get_recovery_evidence(key,first)) - tick() - self.assertEqual(events,[]) - - def test_new_incarnation_abnormal_order_blocks_previous_closure(self): - for kind in ('cpu','service'): - with self.subTest(kind=kind),case() as store: - key, first, *_ = initial_closure(store,kind) - old_id = sql(store,'SELECT id FROM errors')[0][0] - store.acknowledge_error(key,suppression_hours=-1) - store.clear_error(key) - Clock.epoch = BASE-100 - abnormal(store,kind,99) - self.assertGreater(sql(store,'SELECT id FROM errors')[0][0],old_id) - self.assertEqual(sql(store,'SELECT event_type FROM events ORDER BY id DESC')[0][0],'new') - self.assertIsNone(store.get_recovery_evidence(key,first)) - - -if __name__ == '__main__': - unittest.main() diff --git a/AppImage/messages/de/common.json b/AppImage/messages/de/common.json index 424c5aba..f2539122 100644 --- a/AppImage/messages/de/common.json +++ b/AppImage/messages/de/common.json @@ -6266,8 +6266,8 @@ "label": "Neues Gesundheitsproblem" }, "error_resolved": { - "title": "{hostname}: Nicht mehr gemeldet – {category}{entity_suffix}", - "body": "Das Problem in der Kategorie {category} ist nicht mehr in den aktiven Zustandsmeldungen enthalten.\n{reason}\n🚦 Vorheriger Schweregrad: {original_severity}\n⏱️ Zeit seit der ersten Meldung: {duration}", + "title": "{hostname}: Gelöst – {category}{entity_suffix}", + "body": "Das Problem {category} wurde behoben.\n{reason}\n🚦 Vorheriger Schweregrad: {original_severity}\n⏱️ Dauer: {duration}", "label": "Wiederherstellungsbenachrichtigung" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Sicherung gestartet" }, "backup_complete": { - "title": "{hostname}: Backup-Ergebnis nicht bestätigt", - "body": "Das Backup-Ergebnis lässt sich anhand dieser Meldung nicht bestätigen.", + "title": "{hostname} → {storage}: Sicherung abgeschlossen – {vmname} ({vmid})", + "body": "Die Sicherung von {vmname} (ID: {vmid}) wurde am {storage} erfolgreich abgeschlossen.\nGröße: {size}", "label": "Sicherung abgeschlossen" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: Host-Wiederherstellung abgeschlossen", - "body": "Aufgaben nach der Wiederherstellung im Hintergrund abgeschlossen.\n\nÜbernommene Gäste: {guests}\nBind-Mount-Platzhalter: {stubs}\nEntfernte veraltete Knotenverzeichnisse: {stale_nodes}\nNeu installierte Komponenten: {components}\nDauer: {duration}\n{warnings_block}", + "body": "Aufgaben nach der Wiederherstellung im Hintergrund ausgeführt.\n\nGäste haben sich beworben: {guests}\nBind-Mount-Stubs: {stubs}\nVeraltete Knotenverzeichnisse entfernt: {stale_nodes}\nKomponenten neu installiert: {components}\nDauer: {duration}\n{warnings_block}\nDer Knoten ist nun vollständig einsatzbereit.", "label": "Host-Wiederherstellung abgeschlossen" }, "system_problem": { @@ -6910,8 +6910,7 @@ "warning": "Warnung", "info": "Informationen", "ok": "Gelöst", - "default": "Hinweis", - "observation": "Nicht mehr gemeldet" + "default": "Hinweis" }, "groups": { "vm_ct": "Virtuelle Maschine / Container", @@ -6992,8 +6991,8 @@ "temperature": { "sampleSpan": "Die hohen Messwerte erstrecken sich über {duration}." }, - "healthRecovery": {"title": "{hostname}: Behoben - {category}{entity_suffix}", "body": "Eine aktuelle Zustandsprüfung hat für {category} wieder einen normalen Zustand festgestellt.\nVorherige Beobachtung: {reason}\nVorheriger Schweregrad: {original_severity}\nZeit seit der ersten Beobachtung: {duration}", "status": "Behoben"}, "backup": { + "unconfirmedTitle": "{hostname}: Backup-Ergebnis nicht bestätigt", "confirmedTitle": "{hostname}: Backup abgeschlossen", "confirmedBody": "Backup erfolgreich abgeschlossen.", "errorTitle": "{hostname}: Backup-Fehler gemeldet", diff --git a/AppImage/messages/en/common.json b/AppImage/messages/en/common.json index 7d0ca28a..28b9724e 100644 --- a/AppImage/messages/en/common.json +++ b/AppImage/messages/en/common.json @@ -6251,5 +6251,5 @@ "cancel": "Cancel" } }, - "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: No longer reported - {category}{entity_suffix}","body":"The {category} issue is no longer in active health records.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Time since first observation: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname}: Backup outcome unconfirmed","body":"The backup outcome could not be confirmed from this notice.","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or run the post-install menu (option 2) → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Settings > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice","observation":"No longer reported"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed","completed_with_warnings":"Completed with warnings"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"healthRecovery":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"The {category} condition returned to normal in a fresh health check.\nPrevious observation: {reason}\nPrevious severity: {original_severity}\nTime since first observation: {duration}","status":"Resolved"},"backup":{"confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed.","warningTitle":"{hostname}: Backup completed with warnings","warningBody":"Backup completed with warnings.","diagnosticsOmitted":"Additional diagnostic lines or text omitted: {count}. Original report retained."}}} + "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"The {category} issue has been resolved.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Duration: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname} → {storage}: Backup complete — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}\nThe node is now fully ready to use.","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or run the post-install menu (option 2) → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Settings > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed","completed_with_warnings":"Completed with warnings"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"backup":{"unconfirmedTitle":"{hostname}: Backup outcome unconfirmed","confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed.","warningTitle":"{hostname}: Backup completed with warnings","warningBody":"Backup completed with warnings.","diagnosticsOmitted":"Additional diagnostic lines or text omitted: {count}. Original report retained."}}} } diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index 961a228e..f11f208f 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -6266,8 +6266,8 @@ "label": "Nuevo problema de salud" }, "error_resolved": { - "title": "{hostname}: Ya no se informa de {category}{entity_suffix}", - "body": "El problema de {category} ya no figura entre las incidencias de salud activas.\n{reason}\n🚦 Gravedad anterior: {original_severity}\n⏱️ Tiempo desde la primera observación: {duration}", + "title": "{hostname}: Resuelto - {category}{entity_suffix}", + "body": "El problema {category} se ha resuelto.\n{reason}\n🚦 Gravedad anterior: {original_severity}\n⏱️ Duración: {duration}", "label": "Notificación de recuperación" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Backup iniciado" }, "backup_complete": { - "title": "{hostname}: resultado del backup sin confirmar", - "body": "Esta notificación no permite confirmar el resultado del backup.", + "title": "{hostname} → {storage}: Backup completado — {vmname} ({vmid})", + "body": "El backup de {vmname} (ID: {vmid}) se ha completado correctamente en {storage}.\nTamaño: {size}", "label": "Backup completado" }, "backup_warning": { @@ -6556,8 +6556,8 @@ "label": "Reinicio del sistema" }, "system_restore_completed": { - "title": "{hostname}: restauración del host finalizada", - "body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nConfiguraciones de guests aplicadas: {guests}\nDirectorios auxiliares de montajes bind: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}", + "title": "{hostname}: Restauración del host finalizada", + "body": "Tareas posteriores a la restauración completadas en segundo plano.\n\nGuests aplicados: {guests}\nStubs de bind mount: {stubs}\nDirectorios de nodos obsoletos eliminados: {stale_nodes}\nComponentes reinstalados: {components}\nDuración: {duration}\n{warnings_block}\nEl nodo está listo para usarse.", "label": "Restauración del host completada" }, "system_problem": { @@ -6910,8 +6910,7 @@ "warning": "Advertencia", "info": "Información", "ok": "Resuelto", - "default": "Aviso", - "observation": "Ya no se informa" + "default": "Aviso" }, "groups": { "vm_ct": "Máquina virtual / Contenedor", @@ -6992,11 +6991,11 @@ "temperature": { "sampleSpan": "Lecturas altas registradas a lo largo de {duration}." }, - "healthRecovery": {"title": "{hostname}: Resuelto - {category}{entity_suffix}", "body": "Una comprobación reciente confirma que la condición de {category} volvió a la normalidad.\nObservación anterior: {reason}\nGravedad anterior: {original_severity}\nTiempo desde la primera observación: {duration}", "status": "Resuelto"}, "backup": { - "confirmedTitle": "{hostname}: backup completado", + "unconfirmedTitle": "{hostname}: Resultado del backup sin confirmar", + "confirmedTitle": "{hostname}: Backup completado", "confirmedBody": "Backup completado correctamente.", - "errorTitle": "{hostname}: error notificado en el backup", + "errorTitle": "{hostname}: Backup fallido", "errorBody": "El informe del backup contiene un error.", "unconfirmedBody": "El resultado del backup no está confirmado.", "warningTitle": "{hostname}: Backup completado con advertencias", diff --git a/AppImage/messages/fr/common.json b/AppImage/messages/fr/common.json index a0ffa399..0cce8b90 100644 --- a/AppImage/messages/fr/common.json +++ b/AppImage/messages/fr/common.json @@ -6266,8 +6266,8 @@ "label": "Nouveau problème de santé" }, "error_resolved": { - "title": "{hostname} : Plus signalé – {category}{entity_suffix}", - "body": "Le problème {category} ne figure plus parmi les alertes de santé actives.\n{reason}\n🚦 Gravité précédente : {original_severity}\n⏱️ Temps depuis la première observation : {duration}", + "title": "{hostname} : Résolu - {category}{entity_suffix}", + "body": "Le problème {category} a été résolu.\n{reason}\n🚦 Gravité précédente : {original_severity}\n⏱️ Durée : {duration}", "label": "Notification de récupération" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Sauvegarde démarrée" }, "backup_complete": { - "title": "{hostname} : résultat de la sauvegarde non confirmé", - "body": "Cette notification ne permet pas de confirmer le résultat de la sauvegarde.", + "title": "{hostname} → {storage} : Sauvegarde terminée — {vmname} ({vmid})", + "body": "La sauvegarde de {vmname} (ID : {vmid}) s'est terminée avec succès le {storage}.\nTaille : {size}", "label": "Sauvegarde terminée" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname} : restauration de l'hôte terminée", - "body": "Tâches après restauration terminées en arrière-plan.\n\nInvités appliqués : {guests}\nRépertoires de support des montages bind : {stubs}\nRépertoires de nœuds obsolètes supprimés : {stale_nodes}\nComposants réinstallés : {components}\nDurée : {duration}\n{warnings_block}", + "body": "Tâches post-restauration effectuées en arrière-plan.\n\nInvités postulés : {guests}\nTalons de montage liés : {stubs}\nRépertoires de nœuds obsolètes supprimés : {stale_nodes}\nComposants réinstallés : {components}\nDurée : {duration}\n{warnings_block}\nLe nœud est maintenant entièrement prêt à être utilisé.", "label": "Restauration de l'hôte terminée" }, "system_problem": { @@ -6910,8 +6910,7 @@ "warning": "Avertissement", "info": "Informations", "ok": "Résolu", - "default": "Avis", - "observation": "Plus signalé" + "default": "Avis" }, "groups": { "vm_ct": "Machine Virtuelle / Conteneur", @@ -6992,8 +6991,8 @@ "temperature": { "sampleSpan": "Les relevés élevés s'étendent sur {duration}." }, - "healthRecovery": {"title": "{hostname} : Résolu - {category}{entity_suffix}", "body": "Un contrôle récent confirme le retour à la normale de la condition {category}.\nObservation précédente : {reason}\nGravité précédente : {original_severity}\nTemps depuis la première observation : {duration}", "status": "Résolu"}, "backup": { + "unconfirmedTitle": "{hostname} : résultat de la sauvegarde non confirmé", "confirmedTitle": "{hostname} : sauvegarde terminée", "confirmedBody": "Sauvegarde terminée avec succès.", "errorTitle": "{hostname} : erreur signalée lors de la sauvegarde", diff --git a/AppImage/messages/it/common.json b/AppImage/messages/it/common.json index 456091d1..30b699b8 100644 --- a/AppImage/messages/it/common.json +++ b/AppImage/messages/it/common.json @@ -6266,8 +6266,8 @@ "label": "Nuovo problema sanitario" }, "error_resolved": { - "title": "{hostname}: segnalazione non più attiva - {category}{entity_suffix}", - "body": "Il problema relativo a {category} non figura più tra le segnalazioni di salute attive.\n{reason}\n🚦 Gravità precedente: {original_severity}\n⏱️ Tempo dalla prima segnalazione: {duration}", + "title": "{hostname}: risolto - {category}{entity_suffix}", + "body": "Il problema {category} è stato risolto.\n{reason}\n🚦 Gravità precedente: {original_severity}\n⏱️ Durata: {duration}", "label": "Notifica di recupero" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Backup avviato" }, "backup_complete": { - "title": "{hostname}: esito del backup non confermato", - "body": "Questa notifica non consente di confermare l’esito del backup.", + "title": "{hostname} → {storage}: Backup completato — {vmname} ({vmid})", + "body": "Backup di {vmname} (ID: {vmid}) completato con successo su {storage}.\nTaglia: {size}", "label": "Backup completato" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: ripristino dell'host terminato", - "body": "Attività post-ripristino completate in background.\n\nConfigurazioni guest copiate: {guests}\nDirectory di supporto per montaggi bind create: {stubs}\nDirectory obsolete dei nodi rimosse: {stale_nodes}\nComponenti reinstallati: {components}\nDurata: {duration}\n{warnings_block}", + "body": "Attività post-ripristino completate in background.\n\nGli ospiti hanno presentato domanda: {guests}\nStub con montaggio tramite collegamento: {stubs}\nDirectory dei nodi obsolete rimosse: {stale_nodes}\nComponenti reinstallati: {components}\nDurata: {duration}\n{warnings_block}\nIl nodo è ora completamente pronto per l'uso.", "label": "Ripristino dell'host completato" }, "system_problem": { @@ -6910,8 +6910,7 @@ "warning": "Avvertimento", "info": "Informazioni", "ok": "Risolto", - "default": "Avviso", - "observation": "Non più segnalato" + "default": "Avviso" }, "groups": { "vm_ct": "Macchina virtuale/Contenitore", @@ -6992,8 +6991,8 @@ "temperature": { "sampleSpan": "Intervallo dei campioni sopra soglia: {duration}." }, - "healthRecovery": {"title": "{hostname}: Risolto - {category}{entity_suffix}", "body": "Un controllo recente conferma che la condizione {category} è tornata nella norma.\nOsservazione precedente: {reason}\nGravità precedente: {original_severity}\nTempo dalla prima osservazione: {duration}", "status": "Risolto"}, "backup": { + "unconfirmedTitle": "{hostname}: esito del backup non confermato", "confirmedTitle": "{hostname}: backup completato", "confirmedBody": "Backup completato correttamente.", "errorTitle": "{hostname}: errore segnalato nel backup", diff --git a/AppImage/messages/pt/common.json b/AppImage/messages/pt/common.json index 6b731ed2..63c395b0 100644 --- a/AppImage/messages/pt/common.json +++ b/AppImage/messages/pt/common.json @@ -6266,8 +6266,8 @@ "label": "Novo problema de saúde" }, "error_resolved": { - "title": "{hostname}: Já não comunicado – {category}{entity_suffix}", - "body": "O problema de {category} já não consta dos registos de saúde ativos.\n{reason}\n🚦 Gravidade anterior: {original_severity}\n⏱️ Tempo desde a primeira observação: {duration}", + "title": "{hostname}: Resolvido - {category}{entity_suffix}", + "body": "O problema {category} foi resolvido.\n{reason}\n🚦 Gravidade anterior: {original_severity}\n⏱️ Duração: {duration}", "label": "Notificação de recuperação" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Backup iniciado" }, "backup_complete": { - "title": "{hostname}: resultado do backup não confirmado", - "body": "Esta notificação não permite confirmar o resultado do backup.", + "title": "{hostname} → {storage}: Backup concluído — {vmname} ({vmid})", + "body": "Backup de {vmname} (ID: {vmid}) concluído com sucesso em {storage}.\nTamanho: {size}", "label": "Backup concluído" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: restauração do host concluída", - "body": "Tarefas pós-restauro concluídas em segundo plano.\n\nConvidados aplicados: {guests}\nDiretórios auxiliares de montagens bind: {stubs}\nDiretórios de nós obsoletos removidos: {stale_nodes}\nComponentes reinstalados: {components}\nDuração: {duration}\n{warnings_block}", + "body": "Tarefas pós-restauração concluídas em segundo plano.\n\nConvidados inscritos: {guests}\nStubs de montagem de ligação: {stubs}\nDiretórios de nó obsoletos removidos: {stale_nodes}\nComponentes reinstalados: {components}\nDuração: {duration}\n{warnings_block}\nO nó agora está totalmente pronto para uso.", "label": "Restauração do host concluída" }, "system_problem": { @@ -6910,8 +6910,7 @@ "warning": "Aviso", "info": "Informação", "ok": "Resolvido", - "default": "Aviso", - "observation": "Já não comunicado" + "default": "Aviso" }, "groups": { "vm_ct": "Máquina Virtual/Contêiner", @@ -6992,8 +6991,8 @@ "temperature": { "sampleSpan": "As amostras elevadas abrangem {duration}." }, - "healthRecovery": {"title": "{hostname}: Resolvido - {category}{entity_suffix}", "body": "Uma verificação recente confirma que a condição de {category} voltou ao normal.\nObservação anterior: {reason}\nGravidade anterior: {original_severity}\nTempo desde a primeira observação: {duration}", "status": "Resolvido"}, "backup": { + "unconfirmedTitle": "{hostname}: resultado do backup não confirmado", "confirmedTitle": "{hostname}: backup concluído", "confirmedBody": "Backup concluído com sucesso.", "errorTitle": "{hostname}: erro comunicado no backup", diff --git a/AppImage/messages/sv/common.json b/AppImage/messages/sv/common.json index d1842183..3f0f4aa2 100644 --- a/AppImage/messages/sv/common.json +++ b/AppImage/messages/sv/common.json @@ -6266,8 +6266,8 @@ "label": "Nytt hälsoproblem" }, "error_resolved": { - "title": "{hostname}: Rapporteras inte längre – {category}{entity_suffix}", - "body": "Problemet i kategorin {category} finns inte längre bland aktiva hälsoposter.\n{reason}\n🚦 Tidigare allvarlighetsgrad: {original_severity}\n⏱️ Tid sedan första observationen: {duration}", + "title": "{hostname}: Löst - {category}{entity_suffix}", + "body": "{category}-problemet har lösts.\n{reason}\n🚦 Tidigare svårighetsgrad: {original_severity}\n⏱️ Varaktighet: {duration}", "label": "Återställningsmeddelande" }, "error_escalated": { @@ -6396,8 +6396,8 @@ "label": "Säkerhetskopiering startade" }, "backup_complete": { - "title": "{hostname}: säkerhetskopians resultat obekräftat", - "body": "Det går inte att bekräfta säkerhetskopians resultat utifrån denna avisering.", + "title": "{hostname} → {storage}: Säkerhetskopiering klar — {vmname} ({vmid})", + "body": "Säkerhetskopiering av {vmname} (ID: {vmid}) slutfördes framgångsrikt på {storage}.\nStorlek: {size}", "label": "Säkerhetskopieringen är klar" }, "backup_warning": { @@ -6557,7 +6557,7 @@ }, "system_restore_completed": { "title": "{hostname}: Värdåterställning avslutad", - "body": "Åtgärder efter återställning slutfördes i bakgrunden.\n\nGäster tillämpade: {guests}\nHjälpkataloger för bind-monteringar: {stubs}\nFöråldrade nodkataloger borttagna: {stale_nodes}\nKomponenter ominstallerade: {components}\nVaraktighet: {duration}\n{warnings_block}", + "body": "Uppgifter efter återställning slutförda i bakgrunden.\n\nGäster ansökte: {guests}\nBind-monterade stubbar: {stubs}\nInaktuella nodkataloger har tagits bort: {stale_nodes}\nKomponenter installerade om: {components}\nVaraktighet: {duration}\n{warnings_block}\nNoden är nu helt redo att användas.", "label": "Värdåterställning slutförd" }, "system_problem": { @@ -6910,8 +6910,7 @@ "warning": "Varning", "info": "Information", "ok": "Löst", - "default": "Observera", - "observation": "Rapporteras inte längre" + "default": "Observera" }, "groups": { "vm_ct": "Virtuell maskin / behållare", @@ -6992,8 +6991,8 @@ "temperature": { "sampleSpan": "De höga mätvärdena sträcker sig över {duration}." }, - "healthRecovery": {"title": "{hostname}: Åtgärdat - {category}{entity_suffix}", "body": "En aktuell hälsokontroll bekräftar att tillståndet för {category} återgått till det normala.\nTidigare observation: {reason}\nTidigare allvarlighetsgrad: {original_severity}\nTid sedan första observationen: {duration}", "status": "Åtgärdat"}, "backup": { + "unconfirmedTitle": "{hostname}: säkerhetskopians resultat obekräftat", "confirmedTitle": "{hostname}: säkerhetskopiering klar", "confirmedBody": "Säkerhetskopieringen slutfördes utan fel.", "errorTitle": "{hostname}: fel rapporterat vid säkerhetskopiering", diff --git a/AppImage/scripts/build_appimage.sh b/AppImage/scripts/build_appimage.sh index 2e7d6a0e..ccbeff2b 100755 --- a/AppImage/scripts/build_appimage.sh +++ b/AppImage/scripts/build_appimage.sh @@ -131,7 +131,6 @@ cp "$SCRIPT_DIR/auth_manager.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ cp "$SCRIPT_DIR/jwt_middleware.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ jwt_middleware.py not found" cp "$SCRIPT_DIR/health_monitor.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_monitor.py not found" cp "$SCRIPT_DIR/health_persistence.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_persistence.py not found" -cp "$SCRIPT_DIR/health_recovery.py" "$APP_DIR/usr/bin/" cp "$SCRIPT_DIR/flask_health_routes.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ flask_health_routes.py not found" cp "$SCRIPT_DIR/flask_proxmenux_routes.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ flask_proxmenux_routes.py not found" cp "$SCRIPT_DIR/post_install_versions.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ post_install_versions.py not found" diff --git a/AppImage/scripts/health_monitor.py b/AppImage/scripts/health_monitor.py index fdfe931f..26d38470 100644 --- a/AppImage/scripts/health_monitor.py +++ b/AppImage/scripts/health_monitor.py @@ -1427,8 +1427,6 @@ class HealthMonitor: 'details': f'Sustained for {actual_duration}s above {self.CPU_CRITICAL}%.', 'cpu_percent': cpu_percent, 'duration': actual_duration, - 'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL, - 'recovery': self.CPU_RECOVERY}, }, ) elif len(warning_samples) >= WARNING_MIN_SAMPLES and len(recovery_samples) < RECOVERY_MIN_SAMPLES: @@ -1448,32 +1446,13 @@ class HealthMonitor: 'details': f'Sustained for {actual_duration}s above {self.CPU_WARNING}%.', 'cpu_percent': cpu_percent, 'duration': actual_duration, - 'cpu_policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL, - 'recovery': self.CPU_RECOVERY}, }, ) else: status = 'OK' reason = None # CPU is normal - auto-resolve any existing CPU errors - evidence = None - # Presentation proof is stricter than operational hysteresis: - # a supported warning can be below the fixed recovery cutoff. - from health_recovery import _finite_number - criterion = min(self.CPU_WARNING, self.CPU_RECOVERY) - normal_samples = [entry for entry in self.state_history[state_key] - if _finite_number(entry['value']) and 0 <= entry['value'] < criterion - and _finite_number(entry['time']) - and 0 <= current_time - entry['time'] <= self.CPU_RECOVERY_DURATION] - if (_finite_number(cpu_percent) and 0 <= cpu_percent < criterion - and len(normal_samples) >= RECOVERY_MIN_SAMPLES): - evidence = {'check': 'cpu_usage', 'checked_at': current_time, - 'value': cpu_percent, 'normal_samples': len(normal_samples), - 'max_sample': max(entry['value'] for entry in normal_samples), - 'policy': {'warning': self.CPU_WARNING, 'critical': self.CPU_CRITICAL, - 'recovery': self.CPU_RECOVERY}} - health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal', - check_evidence=evidence) + health_persistence.resolve_error('cpu_usage', 'CPU usage returned to normal') temp_status = self._check_cpu_temperature() @@ -3921,7 +3900,6 @@ class HealthMonitor: failed_services = [] service_details = {} - active_evidence = {} for service in services_to_check: try: @@ -3936,10 +3914,6 @@ class HealthMonitor: if result.returncode != 0 or status != 'active': failed_services.append(service) service_details[service] = status or 'inactive' - else: - active_evidence[service] = {'check': f'pve_service_{service}', - 'checked_at': time.time(), 'service': service, 'state': status, - 'returncode': result.returncode} except Exception: failed_services.append(service) service_details[service] = 'error' @@ -3954,7 +3928,7 @@ class HealthMonitor: if svc not in failed_services: error_key = f'pve_service_{svc}' if health_persistence.is_error_active(error_key): - health_persistence.clear_error(error_key, check_evidence=active_evidence.get(svc)) + health_persistence.clear_error(error_key) # Build checks dict with status per service checks = {} diff --git a/AppImage/scripts/health_persistence.py b/AppImage/scripts/health_persistence.py index d06414e2..4fb13fc7 100644 --- a/AppImage/scripts/health_persistence.py +++ b/AppImage/scripts/health_persistence.py @@ -600,7 +600,7 @@ class HealthPersistence: cursor.execute(''' SELECT id, acknowledged, resolved_at, category, severity, first_seen, - notification_sent, suppression_hours, acknowledged_at, details + notification_sent, suppression_hours, acknowledged_at FROM errors WHERE error_key = ? ''', (error_key,)) existing = cursor.fetchone() @@ -609,7 +609,7 @@ class HealthPersistence: if existing: (err_id, ack, resolved_at, old_cat, old_severity, first_seen, - notif_sent, stored_suppression, acknowledged_at, old_details_json) = existing + notif_sent, stored_suppression, acknowledged_at) = existing if ack == 1: # SAFETY OVERRIDE: Critical CPU temperature ALWAYS re-triggers @@ -680,18 +680,6 @@ class HealthPersistence: conn.commit() return event_info - # Original CPU policy is immutable for this row's incident. - # Never upgrade a legacy row from later/current settings. - if error_key == 'cpu_usage': - try: - old_details = json.loads(old_details_json or '{}') - except (ValueError, TypeError): - old_details = {} - details = dict(details) if isinstance(details, dict) else {} - details.pop('cpu_policy', None) - if isinstance(old_details, dict) and 'cpu_policy' in old_details: - details['cpu_policy'] = old_details['cpu_policy'] - details_json = json.dumps(details) # Not acknowledged - update existing active error cursor.execute(''' UPDATE errors @@ -756,12 +744,12 @@ class HealthPersistence: return event_info - def resolve_error(self, error_key: str, reason: str = 'auto-resolved', *, check_evidence=None): + def resolve_error(self, error_key: str, reason: str = 'auto-resolved'): """Mark an error as resolved""" with self._db_lock: - return self._resolve_error_impl(error_key, reason, check_evidence=check_evidence) + return self._resolve_error_impl(error_key, reason) - def _resolve_error_impl(self, error_key, reason, *, check_evidence=None): + def _resolve_error_impl(self, error_key, reason): with self._db_connection() as conn: cursor = conn.cursor() now = datetime.now().isoformat() @@ -788,7 +776,7 @@ class HealthPersistence: # was created — otherwise "Storage 'Tuxis' unavailable" # comes back as "Resolved - Storage" with no identity. cursor.execute( - 'SELECT details, id, first_seen, resolved_at FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1', + 'SELECT details FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1', (error_key,), ) row = cursor.fetchone() @@ -798,71 +786,14 @@ class HealthPersistence: stored_details = json.loads(row[0]) except Exception: stored_details = None - # Legacy/mismatched original policy cannot certify normality. - if error_key == 'cpu_usage' and (not isinstance(stored_details, dict) - or not isinstance(check_evidence, dict) - or check_evidence.get('policy') != stored_details.get('cpu_policy') - or not stored_details.get('cpu_policy')): - check_evidence = None self._record_event(cursor, 'resolved', error_key, { 'reason': reason, - # Only explicit current-check callers attach this proof. - # Generic resolve/cleanup remains neutral. - 'check_evidence': check_evidence, - 'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': row[3]}, 'entity': self._entity_from_details(stored_details), 'details': stored_details or {}, }) conn.commit() - def get_recovery_evidence(self, error_key: str, first_seen: str): - """Return fresh same-incident native check proof, never absence of errors. - - Host CPU and exact per-service active checks carry provenance. Other - checks, generic clears, excluded/deleted records and legacy events stay - neutral until they have equivalent per-condition provenance. - """ - if not isinstance(error_key, str) or not first_seen or not ( - error_key == 'cpu_usage' or error_key.startswith('pve_service_')): - return None - try: - # One SQLite statement is one consistent row/ack/closure snapshot. - # Latest native observation/closure by durable event id: a later - # abnormal record supersedes proof even when a resolved row is - # reused and wall-clock time moves backward. No-op clears create - # no event, so they do not invalidate a genuine closure. - with self._db_lock, self._db_connection() as conn: - row = conn.execute(''' - SELECT e.first_seen, e.last_seen, e.resolved_at, e.acknowledged, - e.id, v.timestamp, v.data - FROM errors e JOIN events v ON v.id = ( - SELECT id FROM events WHERE error_key = e.error_key - AND event_type IN ('resolved', 'cleared', 'new', 'updated', 'escalated') - ORDER BY id DESC LIMIT 1 - ) WHERE e.error_key = ? - ''', (error_key,)).fetchone() - if not row or row[0] != first_seen or not row[2] or row[3]: - return None - event_data = json.loads(row[6]) - if event_data.get('incident') != {'id': row[4], 'first_seen': row[0], 'resolved_at': row[2]}: - return None - proof = event_data.get('check_evidence') - from health_recovery import valid_check_evidence - if not valid_check_evidence(error_key, proof, now=datetime.now().timestamp()): - return None - if error_key == 'cpu_usage' and proof.get('policy') != event_data.get('details', {}).get('cpu_policy'): - return None - checked = float(proof['checked_at']) - last_seen = datetime.fromisoformat(row[1]).timestamp() - resolved = datetime.fromisoformat(row[2]).timestamp() - recorded = datetime.fromisoformat(row[5]).timestamp() - if not last_seen <= checked <= resolved <= recorded: - return None - return proof - except (ValueError, TypeError, AttributeError, OverflowError): - return None - def is_error_active(self, error_key: str, category: Optional[str] = None) -> bool: """ Check if an error is currently active OR suppressed (dismissed but within suppression period). @@ -928,7 +859,7 @@ class HealthPersistence: return False - def clear_error(self, error_key: str, *, check_evidence=None): + def clear_error(self, error_key: str): """ Remove/resolve a specific error immediately. Used when the condition that caused the error no longer exists @@ -950,7 +881,7 @@ class HealthPersistence: # Check if this error was acknowledged (dismissed) cursor.execute(''' - SELECT acknowledged, id, first_seen FROM errors WHERE error_key = ? + SELECT acknowledged FROM errors WHERE error_key = ? ''', (error_key,)) row = cursor.fetchone() @@ -969,10 +900,7 @@ class HealthPersistence: ''', (now, error_key)) if cursor.rowcount > 0: - self._record_event(cursor, 'cleared', error_key, { - 'reason': 'condition_resolved', 'check_evidence': check_evidence, - 'incident': {'id': row[1], 'first_seen': row[2], 'resolved_at': now}, - }) + self._record_event(cursor, 'cleared', error_key, {'reason': 'condition_resolved'}) conn.commit() diff --git a/AppImage/scripts/health_recovery.py b/AppImage/scripts/health_recovery.py deleted file mode 100644 index 9a8a00df..00000000 --- a/AppImage/scripts/health_recovery.py +++ /dev/null @@ -1,51 +0,0 @@ -"""Bounded recovery metadata admission, without importing monitor singletons. - -Native persistence additionally binds the exact incident and closure. Manual -notifications remain authenticated caller assertions: shape validation cannot -establish that an asserted measurement actually happened. -""" -import math -import time -from typing import TypeGuard - - -def _finite_number(value) -> TypeGuard[int | float]: - try: - return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value) - except (OverflowError, ValueError, TypeError): - return False - - -def valid_check_evidence(error_key, proof, *, now=None): - """Validate a supported measurement contract, not its external authenticity.""" - if not isinstance(proof, dict) or proof.get('check') != error_key: - return False - checked = proof.get('checked_at') - now = time.time() if now is None else now - if not _finite_number(checked) or not _finite_number(now) or not 0 <= now-checked <= 7200: - return False - if error_key == 'cpu_usage': - policy = proof.get('policy') - if not isinstance(policy, dict) or set(policy) != {'warning', 'critical', 'recovery'}: - return False - if not all(_finite_number(v) and 1 <= v <= 100 for v in policy.values()): - return False - if policy['warning'] > policy['critical']: - return False - value, maximum, count = proof.get('value'), proof.get('max_sample'), proof.get('normal_samples') - return (_finite_number(value) and _finite_number(maximum) - and 0 <= value <= maximum < min(policy['warning'], policy['recovery']) - and isinstance(count, int) and not isinstance(count, bool) and count >= 10) - if isinstance(error_key, str) and error_key.startswith('pve_service_'): - service = error_key[len('pve_service_'):] - return (bool(service) and proof.get('service') == service - and proof.get('state') == 'active' - and type(proof.get('returncode')) is int and proof['returncode'] == 0) - return False - - -def presents_recovery(data): - """One presentation predicate shared by template, icon and email badge.""" - return (isinstance(data, dict) and data.get('recovery_outcome') == 'resolved' - and data.get('is_recovery') is True - and valid_check_evidence(data.get('error_key'), data.get('check_evidence'))) diff --git a/AppImage/scripts/notification_channels.py b/AppImage/scripts/notification_channels.py index 584ad7a1..c2bff230 100644 --- a/AppImage/scripts/notification_channels.py +++ b/AppImage/scripts/notification_channels.py @@ -1037,15 +1037,7 @@ class EmailChannel(NotificationChannel): # Determine group for section header event_type = data.get('_event_type', '') - if event_type == 'error_resolved': - from health_recovery import presents_recovery - if presents_recovery(data): - sev.update(self._SEV_STYLE['OK']) - sev['label'] = _runtime_notification_text('healthRecovery.status', data) - else: - sev.update(self._SEV_DEFAULT) - sev['label'] = _runtime_text('email.severity.observation', data) - elif event_type == 'backup_complete': + if event_type == 'backup_complete': outcome = data.get('backup_outcome') if outcome == 'confirmed': sev.update(self._SEV_STYLE['OK']) @@ -1064,13 +1056,12 @@ class EmailChannel(NotificationChannel): # Scoped inline mail-compatible wrapping for authoritative raw-context # bodies, including restore bodies released from quiet hours. backup_email = event_type in {'backup_complete', 'backup_fail'} - wrap_body = (event_type in {'temp_high', 'system_restore_completed', 'error_resolved'} + wrap_body = (event_type in {'temp_high', 'system_restore_completed'} or backup_email or data.get('_restore_summary') or data.get('_backup_summary')) temp_cell_wrap = 'word-wrap:break-word;overflow-wrap:break-word;word-break:break-word;' if wrap_body else '' temp_table_layout = 'table-layout:fixed;' if wrap_body else '' - # Recovery exposes the same literal host context as backup notices. # Keep wrapping event-scoped; unrelated mail remains byte-identical. - context_email = backup_email or event_type == 'error_resolved' + context_email = backup_email backup_title_wrap = temp_cell_wrap if context_email else '' backup_metadata_layout = 'table-layout:fixed;' if context_email else '' section_label = _runtime_text(f'email.groups.{group}', data) @@ -1106,12 +1097,11 @@ class EmailChannel(NotificationChannel): # A metadata-only/manual body may be generic. Keep actionable raw # context once, without restoring duplicated inventory metadata. reason = data.get('reason', '') - if reason and len(reason) <= 80 and reason not in body: + if backup_email and reason and len(reason) <= 80 and reason not in body: detail_rows.append((html_mod.escape(_runtime_text('email.fields.reason', data)), html_mod.escape(reason))) - if event_type in {'system_restore_completed', 'error_resolved'} or data.get('_restore_summary'): - # Observation age/disappearance must not become a green OK row. + if event_type == 'system_restore_completed' or data.get('_restore_summary'): # The endpoint's warnings_block and task counts live in the # localized body, not the generic services Event row. detail_rows = [('', html_mod.escape(line if data.get('_quiet_hours_summary') else line.strip())) @@ -1150,7 +1140,7 @@ class EmailChannel(NotificationChannel): reason = data.get('reason', '') reason_html = '' if reason and len(reason) > 80 and not ( - (event_type in {'temp_high', 'error_resolved'} or backup_email) and reason in body): + (event_type == 'temp_high' or backup_email) and reason in body): reason_html = f'''

{_runtime_text('email.details', data)}

diff --git a/AppImage/scripts/notification_events.py b/AppImage/scripts/notification_events.py index 9aedaf3a..9272ff69 100644 --- a/AppImage/scripts/notification_events.py +++ b/AppImage/scripts/notification_events.py @@ -3223,13 +3223,6 @@ class PollingCollector: self._last_notified.pop(key, None) continue - # Disappearance is not recovery. Only same-incident proof written - # by a successful existing native check can certify normality. - try: - recovery_evidence = health_persistence.get_recovery_evidence(key, first_seen) - except Exception: - recovery_evidence = None - # Calculate duration duration = '' if first_seen: @@ -3261,7 +3254,7 @@ class PollingCollector: reason_lines = (reason or '').split('\n') reason_summary = reason_lines[0] if reason_lines else '' - # Keep the earlier device context without asserting recovery. + # Try to extract device info for a clean "Device: xxx (recovered)" line device_line = '' for line in reason_lines: if 'Device:' in line or 'Device not currently' in line or '/dev/' in line: @@ -3274,15 +3267,12 @@ class PollingCollector: break if reason_summary and device_line: - clean_reason = f'{reason_summary}\n{device_line} (no longer reported)' + clean_reason = f'{reason_summary}\n{device_line} (recovered)' elif reason_summary: - clean_reason = f'{reason_summary} (no longer reported)' + clean_reason = f'{reason_summary} (recovered)' else: - clean_reason = 'Condition no longer reported' + clean_reason = 'Condition resolved' - if recovery_evidence: - clean_reason = reason_summary - # `original_severity` must match what the user actually saw # in the most-recent notification for this error, not the # latest DB severity. See `_notified_severity` docstring at @@ -3309,9 +3299,7 @@ class PollingCollector: 'original_severity': original_severity, 'first_seen': first_seen, 'duration': duration_label, - 'is_recovery': bool(recovery_evidence), - 'recovery_outcome': 'resolved' if recovery_evidence else 'no_longer_reported', - 'check_evidence': recovery_evidence, + 'is_recovery': True, } # Spread the original details blob so the resolved notification # can use the same {storage_name}/{vm_name}/{device} placeholders diff --git a/AppImage/scripts/notification_manager.py b/AppImage/scripts/notification_manager.py index 56abc814..178bd087 100644 --- a/AppImage/scripts/notification_manager.py +++ b/AppImage/scripts/notification_manager.py @@ -2012,7 +2012,9 @@ class NotificationManager: 'digest.quietTitle', language, hostname=host, count=len(rows), ) use_icons = self._config.get(f'{ch_name}.rich_format', 'false') == 'true' - summary_body = self._compose_digest_body(rows, use_icons=use_icons, quiet_release=True) + quiet_details = any(row[1] in ('backup_complete', 'backup_fail', 'system_restore_completed') + for row in rows) + summary_body = self._compose_digest_body(rows, use_icons=use_icons, quiet_release=quiet_details) result: dict = {'success': False, 'error': ''} try: @@ -2498,7 +2500,7 @@ class NotificationManager: runtime_data.setdefault('_notification_language', self._notification_language()) # Match queued dispatch's presentation context for these outcome # notices; this does not alter event/severity or direct-send policy. - if event_type in ('backup_complete', 'backup_fail', 'error_resolved', 'system_restore_completed'): + if event_type in ('backup_complete', 'backup_fail', 'system_restore_completed'): runtime_data['_event_type'] = event_type runtime_data['_group'] = TEMPLATES[event_type].get('group', 'other') diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index f0d055b4..2192c50f 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -817,10 +817,10 @@ TEMPLATES = { # `{entity}` is populated by health_persistence.resolve_error() # (via _entity_from_details) and by PollingCollector's spread of # the original details blob. When absent, _SafeDict elides the - # placeholder and the title collapses back to "No longer reported - " + # placeholder and the title collapses back to "Resolved - " # without a trailing dash. - 'title': '{hostname}: No longer reported - {category}{entity_suffix}', - 'body': 'The {category} issue is no longer in active health records.\n{reason}\n\U0001F6A6 Previous severity: {original_severity}\n\u23F1\uFE0F Time since first observation: {duration}', + 'title': '{hostname}: Resolved - {category}{entity_suffix}', + 'body': 'The {category} issue has been resolved.\n{reason}\n\U0001F6A6 Previous severity: {original_severity}\n\u23F1\uFE0F Duration: {duration}', 'label': 'Recovery notification', 'group': 'health', 'default_enabled': True, @@ -1037,8 +1037,8 @@ TEMPLATES = { 'default_enabled': False, }, 'backup_complete': { - 'title': '{hostname}: Backup outcome unconfirmed', - 'body': 'The backup outcome could not be confirmed from this notice.', + 'title': '{hostname} → {storage}: Backup complete — {vmname} ({vmid})', + 'body': 'Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}', 'label': 'Backup complete', 'group': 'backup', 'default_enabled': True, @@ -1308,7 +1308,8 @@ TEMPLATES = { 'Stale node dirs removed: {stale_nodes}\n' 'Components reinstalled: {components}\n' 'Duration: {duration}\n' - '{warnings_block}' + '{warnings_block}\n' + 'The node is now fully ready to use.' ), 'label': 'Host restore completed', 'group': 'services', @@ -1894,6 +1895,10 @@ def render_template(event_type: str, data: Dict[str, Any], template[field] = localized backup_title_target = '' if event_type == 'backup_complete': + template['title'] = runtime_message('backup.unconfirmedTitle', language, + hostname=data.get('hostname') or _get_hostname()) or ( + str(data.get('hostname') or _get_hostname()) + ': Backup outcome unconfirmed') + template['body'] = runtime_message('backup.unconfirmedBody', language) or 'The backup outcome is not confirmed.' outcome = data.get('backup_outcome') if outcome == 'confirmed': template['title'] = runtime_message('backup.confirmedTitle', language, @@ -1961,7 +1966,7 @@ def render_template(event_type: str, data: Dict[str, Any], 'log_file': '', } variables.update(data) - if event_type == 'backup_fail' or (event_type == 'backup_complete' and data.get('backup_outcome') in ('confirmed', 'completed_with_warnings', 'failed')): + if event_type in ('backup_fail', 'backup_complete'): # The provider has already substituted raw Display Names. Insert the # resolved title as a value, never reinterpret its literal braces. variables['_backup_title'] = template['title'] @@ -2070,13 +2075,6 @@ def render_template(event_type: str, data: Dict[str, Any], return '' safe_vars = _SafeDict(variables) - if event_type == 'error_resolved': - from health_recovery import presents_recovery - if presents_recovery(data): - safe_vars['_health_title'] = runtime_message('healthRecovery.title', language, **variables) - safe_vars['_health_body'] = runtime_message('healthRecovery.body', language, **variables) - template['title'] = '{_health_title}' - template['body'] = '{_health_body}' try: title = template['title'].format_map(safe_vars) except (ValueError, IndexError): @@ -2152,7 +2150,7 @@ def render_template(event_type: str, data: Dict[str, Any], key = ('backup.errorBody' if data.get('backup_outcome') == 'failed' else 'backup.warningBody' if data.get('backup_outcome') == 'completed_with_warnings' else 'backup.unconfirmedBody') - body_text = runtime_message(key, language) + '\n' + body_text + body_text = (runtime_message(key, language) or template['body']) + '\n' + body_text elif event_type == 'system_mail' and pve_message: # System mail -- use PVE message directly (mail bounce, cron, smartd) body_text = pve_message.strip()[:1000] @@ -2181,14 +2179,19 @@ def render_template(event_type: str, data: Dict[str, Any], if event_type in ('backup_complete', 'backup_fail') and ( event_type == 'backup_fail' or data.get('backup_outcome') == 'failed'): source_subject = str(data.get('pve_title') or '').strip() + native_failure = re.fullmatch( + r'vzdump backup status \([^\r\n]*\): backup failed(?::\s*(.*))?', + source_subject, re.IGNORECASE) guest_context = (_parse_vzdump_message(str(pve_message or '')) or {}).get('vms') - if guest_context: - # Native single-line job errors live only in the subject. Retain - # that cause, not the redundant job/host envelope or generic count. + if native_failure: + # Before the first guest, too, only the native cause is diagnostic; + # the original host/job envelope is not display-name context. + source_subject = (native_failure.group(1) or '').strip() + elif guest_context: cause = re.search(r'\bbackup failed:\s*(.+)', source_subject, re.IGNORECASE) source_subject = cause.group(1).strip() if cause else '' - if source_subject.lower() == 'multiple problems': - source_subject = '' + if source_subject.lower() == 'multiple problems': + source_subject = '' if source_subject and source_subject not in {line.strip() for line in body_text.splitlines()}: # Reserve the subject-equivalent diagnostic BEFORE the cap. Finding # it in uncapped logs is not enough: that late line could be omitted. @@ -2374,14 +2377,14 @@ EVENT_EMOJI = { 'system_startup': '\U0001F680', # rocket (startup) 'system_shutdown': '\u23FB\uFE0F', # power symbol (Unicode) 'system_reboot': '\U0001F504', - 'system_restore_completed': '\U0001F4CB', # post-restore task report (boot may have warnings) + 'system_restore_completed': '✅', # check mark 'system_problem': '\u26A0\uFE0F', 'kernel_warning': '\u26A0\uFE0F', 'service_fail': '\u274C', 'oom_kill': '\U0001F4A3', # bomb # Health 'new_error': '\U0001F198', # SOS - 'error_resolved': '\U0001F4CB', # no longer active in health records, not proven recovery + 'error_resolved': '\u2705', 'error_escalated': '\U0001F53A', # red triangle up 'health_degraded': '\u26A0\uFE0F', 'health_persistent': '\U0001F4CB', # clipboard @@ -2533,10 +2536,6 @@ def enrich_with_emojis(event_type: str, title: str, body: str, severity = data.get('severity', 'INFO') icon = EVENT_EMOJI.get(event_type) or CATEGORY_EMOJI.get(group) or SEVERITY_ICONS.get(severity, '') - if event_type == 'error_resolved': - from health_recovery import presents_recovery - if presents_recovery(data): - icon = '✅' if event_type == 'backup_complete': icon = { 'confirmed': '💾✅', 'completed_with_warnings': '💾⚠️', 'failed': '💾❌', diff --git a/AppImage/scripts/tests/test_notification_runtime_i18n.py b/AppImage/scripts/tests/test_notification_runtime_i18n.py index cf98806c..58961d09 100644 --- a/AppImage/scripts/tests/test_notification_runtime_i18n.py +++ b/AppImage/scripts/tests/test_notification_runtime_i18n.py @@ -50,10 +50,6 @@ class RuntimeCatalogTests(unittest.TestCase): self.assertIsInstance(templates[event_type][field], str) self.assertTrue(templates[event_type][field]) if field in source: - # Upstream Slovak backup title/body still belong to - # the pre-outcome schema; runtime falls back to EN. - if language == 'sk' and event_type == 'backup_complete' and field != 'label': - continue self.assertEqual( _placeholders(templates[event_type][field]), _placeholders(source[field]), @@ -74,10 +70,9 @@ class RuntimeCatalogTests(unittest.TestCase): en = flatten(self.catalogs["en"]) pending_slovak = {"backup.confirmedTitle", "backup.confirmedBody", "backup.errorTitle", "backup.errorBody", "backup.unconfirmedBody", - "channels.email.severity.observation", "channels.email.status.unconfirmed", + "backup.unconfirmedTitle", "channels.email.status.unconfirmed", "backup.warningTitle", "backup.warningBody", "backup.diagnosticsOmitted", - "channels.email.status.completed_with_warnings", - "healthRecovery.title", "healthRecovery.body", "healthRecovery.status"} + "channels.email.status.completed_with_warnings"} for language, catalog in self.catalogs.items(): translated = flatten(catalog) if language == 'sk': @@ -90,9 +85,6 @@ class RuntimeCatalogTests(unittest.TestCase): for key in translated: self.assertIsInstance(translated[key], str, f"{language}:{key}") self.assertTrue(translated[key].strip(), f"{language}:{key}") - if language == 'sk' and key in ('templates.backup_complete.title', - 'templates.backup_complete.body'): - continue # exact upstream SK, superseded only at render time self.assertEqual(_placeholders(translated[key]), _placeholders(en[key]), f"{language}:{key}") def test_notification_language_ui_keys_exist_in_both_catalogs(self): @@ -132,6 +124,11 @@ class RuntimeCatalogTests(unittest.TestCase): with mock.patch.object(notification_templates, "_get_hostname", return_value="HOST-ŽILINA"): for event_type in notification_templates.TEMPLATES: event_values = dict(values) + if event_type == "backup_complete": + # Legacy catalog fields keep their completion meaning; + # the new unknown-outcome keys intentionally omit guest data. + # Exercise retained metadata on the explicit confirmed path. + event_values["backup_outcome"] = "confirmed" if event_type == "temp_high": # Temperature is a measured numeric contract; arbitrary # DYNAMIC_VALUE is correctly rejected by its fallback.