From 41ae752da14486ad1b03ae9ff07b930304c276bd Mon Sep 17 00:00:00 2001 From: MacRimi Date: Tue, 6 Oct 2026 23:18:26 +0200 Subject: [PATCH] feat(oci): watchdog that restarts an application when it stops on its own - New service that starts again an OCI application that stopped on its own, such as Frigate after Save & Restart. A stop or shutdown asked by the user, a backup, a migration and a ProxMenux operation are never undone. - The installer asks it in both modes and proposes what the recipe declares; it is changed from the management menu or with the Watchdog switch of the container modal in the Monitor. - Notifications when the watchdog restarts an application and when it keeps stopping. - The service is recorded in the change journal, and the feature is documented. --- .../tests/test_oci_operation_events.py | 3 +- .../tests/test_oci_selection_setup_wording.py | 2 +- .../scripts/tests/test_oci_watchdog_wiring.py | 80 +++++ AppImage/components/virtual-machines.tsx | 55 +++- AppImage/messages/de/common.json | 14 + AppImage/messages/en/common.json | 6 +- AppImage/messages/es/common.json | 14 + AppImage/messages/fr/common.json | 14 + AppImage/messages/it/common.json | 14 + AppImage/messages/pt/common.json | 14 + AppImage/messages/sk/common.json | 14 + AppImage/messages/sv/common.json | 14 + AppImage/scripts/flask_notification_routes.py | 3 +- AppImage/scripts/flask_server.py | 33 ++ AppImage/scripts/notification_templates.py | 20 ++ AppImage/scripts/oci_instance_info.py | 7 +- install_proxmenux.sh | 1 + install_proxmenux_beta.sh | 1 + lang/es.json | 9 + oci/remote/oci_watchdog.py | 305 ++++++++++++++++++ oci/src/proxmenux_oci/cli.py | 22 ++ oci/src/proxmenux_oci/management.py | 56 +++- oci/tests/test_stack_recreate_v1.py | 1 + oci/tests/test_watchdog.py | 209 ++++++++++++ .../en/docs/oci-manager/lifecycle.json | 53 ++- web/messages/en/docs/oci-manager/monitor.json | 4 + .../es/docs/oci-manager/lifecycle.json | 53 ++- web/messages/es/docs/oci-manager/monitor.json | 4 + 28 files changed, 1006 insertions(+), 19 deletions(-) create mode 100644 .github/scripts/tests/test_oci_watchdog_wiring.py create mode 100644 oci/remote/oci_watchdog.py create mode 100644 oci/tests/test_watchdog.py diff --git a/.github/scripts/tests/test_oci_operation_events.py b/.github/scripts/tests/test_oci_operation_events.py index 6449914f..32d024df 100644 --- a/.github/scripts/tests/test_oci_operation_events.py +++ b/.github/scripts/tests/test_oci_operation_events.py @@ -7,6 +7,7 @@ from unittest import TestCase ROOT = Path(__file__).resolve().parents[3] LOCALES = ('en', 'es', 'de', 'fr', 'it', 'pt', 'sk', 'sv') EVENTS = [f'oci_{kind}_{result}' for kind in ('update', 'modify', 'recreate') for result in ('completed', 'failed')] +EVENTS += ['oci_watchdog_restarted', 'oci_watchdog_failed'] class OciOperationEvents(TestCase): @@ -17,7 +18,7 @@ class OciOperationEvents(TestCase): for event in EVENTS: self.assertIn(f"'{event}':", accepted, event) self.assertEqual(len(re.findall(rf"^ '{event}': \{{$", templates, re.MULTILINE)), 1, event) - self.assertEqual(len(re.findall(rf"^ '{event}': +'\\u", templates, re.MULTILINE)), 1, event) + self.assertEqual(len(re.findall(rf"^ '{event}': +'\\[uU]", templates, re.MULTILINE)), 1, event) def test_every_language_has_the_label_and_the_message_of_every_event(self): fields = lambda text: sorted(re.findall(r'\{(\w+)\}', text)) diff --git a/.github/scripts/tests/test_oci_selection_setup_wording.py b/.github/scripts/tests/test_oci_selection_setup_wording.py index 3e6df1e1..af2b9d38 100644 --- a/.github/scripts/tests/test_oci_selection_setup_wording.py +++ b/.github/scripts/tests/test_oci_selection_setup_wording.py @@ -69,7 +69,7 @@ class SelectionSetupWording(TestCase): # It cannot be updated, but it can still be changed or removed. ui = self._stack({'members': [{'native_stack_intent': {'adapt': True}}]}, REPLAY) ui.choose.assert_called_once() - self.assertEqual([tag for tag, _ in ui.choose.call_args.args[1]], ['modify', 'remove']) + self.assertEqual([tag for tag, _ in ui.choose.call_args.args[1]], ['modify', 'watchdog', 'remove']) def _stack(self, stack, expected): record = {'stack': stack} diff --git a/.github/scripts/tests/test_oci_watchdog_wiring.py b/.github/scripts/tests/test_oci_watchdog_wiring.py new file mode 100644 index 00000000..a8fb4c49 --- /dev/null +++ b/.github/scripts/tests/test_oci_watchdog_wiring.py @@ -0,0 +1,80 @@ +"""The watchdog of an OCI application is reachable from the Monitor and worded in every language.""" +import ast +import json +from pathlib import Path +import re +from unittest import TestCase + +ROOT = Path(__file__).resolve().parents[3] +LOCALES = ('en', 'es', 'de', 'fr', 'it', 'pt', 'sk', 'sv') + + +class OciWatchdogWiring(TestCase): + def test_changing_it_needs_an_administrator_and_goes_through_the_engine(self): + server = (ROOT / 'AppImage/scripts/flask_server.py').read_text() + route = re.search(r"@app\.route\('/api/lxc//oci-watchdog', methods=\['POST'\]\)\n@(\w+)\ndef (\w+)", server) + self.assertEqual(route.group(1), 'require_admin_scope') + body = server[route.end():server.index("@app.route('/api/vms//logs'")] + self.assertIn("if not isinstance(enabled, bool):", body) + self.assertIn("oci/engine/remote/oci_watchdog.py',", body) + self.assertNotIn('shell=True', body) + self.assertIn('"watchdog": False,', (ROOT / 'AppImage/scripts/oci_instance_info.py').read_text()) + + def test_the_toggle_is_only_offered_for_an_oci_application(self): + modal = (ROOT / 'AppImage/components/virtual-machines.tsx').read_text() + block = modal[modal.index('{/* Watchdog of an OCI application'):] + self.assertTrue(block.split('\n', 3)[2].strip().startswith('{ociInstance?.oci_instance && (')) + self.assertIn('t("vmLxc.details.watchdog")', block) + self.assertIn('`/api/lxc/${selectedVM.vmid}/oci-watchdog`', modal) + + def test_every_language_names_and_explains_it(self): + for locale in LOCALES: + details = json.loads((ROOT / f'AppImage/messages/{locale}/common.json').read_text())['vmLxc']['details'] + self.assertTrue(details['watchdog'].strip(), locale) + self.assertTrue(details['watchdogHelp'].strip(), locale) + + def test_the_engine_restarts_its_service_when_proxmenux_replaces_it(self): + for installer in ('install_proxmenux.sh', 'install_proxmenux_beta.sh'): + self.assertIn('systemctl try-restart proxmenux-oci-watchdog.service', (ROOT / installer).read_text(), installer) + engine = (ROOT / 'oci/remote/oci_watchdog.py').read_text() + self.assertIn("UNIT = 'proxmenux-oci-watchdog.service'", engine) + + +class OciWatchdogInstaller(TestCase): + CLI = (ROOT / 'oci/src/proxmenux_oci/cli.py').read_text() + MENU = (ROOT / 'oci/src/proxmenux_oci/management.py').read_text() + + def default(self, template): + node = next(item for item in ast.parse(self.CLI).body + if isinstance(item, ast.FunctionDef) and item.name == 'watchdog_default') + scope = {'Any': object} + exec(compile(ast.Module(body=[node], type_ignores=[]), 'cli.py', 'exec'), scope) + return scope['watchdog_default'](template) + + def test_the_recipe_decides_what_is_proposed(self): + self.assertTrue(self.default({'container_contract': {'restart': 'unless-stopped'}})) + self.assertTrue(self.default({'container_contract': {'restart': 'always'}})) + self.assertTrue(self.default({'compose_stack': {'services': [{'compose': {}}, {'compose': {'restart': 'on-failure'}}]}})) + self.assertFalse(self.default({'container_contract': {'restart': None}})) + self.assertFalse(self.default({'container_contract': {'restart': 'no'}, 'compose_stack': {'services': [{'compose': {'restart': 'no'}}]}})) + self.assertFalse(self.default({})) + + def test_every_installation_asks_it_and_the_engine_never_receives_the_answer(self): + install = self.CLI[self.CLI.index('def install_template('):self.CLI.index('def _rclone_mount(')] + asked = install.index('deployment["watchdog"] = wizard.confirm(') + self.assertLess(install.index('deployment = build_deployment(candidate, wizard, mode)'), asked) + self.assertLess(asked, install.index('approved = wizard.review(')) + self.assertLess(install.index('watchdog = bool(deployment.pop("watchdog", False))'), install.index('run_remote_install(')) + self.assertIn('set_watchdog(PROJECT_ROOT, sorted(vmids), True)', install) + self.assertIn('row(translate("Watchdog"), _yes_no(plan["watchdog"]))', self.CLI) + + def test_both_management_menus_offer_it(self): + self.assertEqual(self.MENU.count("('watchdog', "), 2) + self.assertEqual(self.MENU.count("if action == 'watchdog':"), 2) + + def test_the_service_is_written_through_the_change_journal(self): + engine = (ROOT / 'oci/remote/oci_watchdog.py').read_text() + self.assertIn('pmx_write_file "$2" && systemctl daemon-reload && pmx_enable_service "$3"', engine) + journal = (ROOT / 'scripts/global/pmx_journal.sh').read_text() + for helper in ('pmx_journal_context()', 'pmx_write_file()', 'pmx_enable_service()'): + self.assertIn(helper, journal) diff --git a/AppImage/components/virtual-machines.tsx b/AppImage/components/virtual-machines.tsx index 722c72ae..b56cebc2 100644 --- a/AppImage/components/virtual-machines.tsx +++ b/AppImage/components/virtual-machines.tsx @@ -52,6 +52,7 @@ interface OciInstanceInfo { host_directories: boolean pending: boolean restored?: boolean + watchdog?: boolean } interface LxcUpdateCheck { @@ -887,6 +888,8 @@ export function VirtualMachines() { // savedOnboot → 2 s ack pill after successful save. const [resourcesEditMode, setResourcesEditMode] = useState(false) const [pendingOnboot, setPendingOnboot] = useState(null) + // Watchdog of an OCI application: same edit-gate as the toggle above. + const [pendingWatchdog, setPendingWatchdog] = useState(null) const [pendingTags, setPendingTags] = useState(null) const [newTagDraft, setNewTagDraft] = useState("") // When set (in edit mode), the pill at this index renders as an @@ -1251,6 +1254,7 @@ export function VirtualMachines() { // start in view mode with no pending change. setResourcesEditMode(false) setPendingOnboot(null) + setPendingWatchdog(null) setPendingTags(null) setNewTagDraft("") setEditingTagIndex(null) @@ -1682,6 +1686,7 @@ export function VirtualMachines() { const handleCancelResourcesEdit = () => { setPendingOnboot(null) + setPendingWatchdog(null) setPendingTags(null) setNewTagDraft("") setEditingTagIndex(null) @@ -1702,16 +1707,26 @@ export function VirtualMachines() { if (pendingTags !== null && stringifyTags(pendingTags) !== stringifyTags(currentTags)) { payload.tags = pendingTags } - if (Object.keys(payload).length === 0) { + const watchdogChanged = !!ociInstance?.oci_instance && pendingWatchdog !== null && pendingWatchdog !== !!ociInstance.watchdog + if (Object.keys(payload).length === 0 && !watchdogChanged) { handleCancelResourcesEdit() return } setSavingOnboot(true) try { - await fetchApi(`/api/vms/${selectedVM.vmid}/config`, { - method: "POST", - body: JSON.stringify(payload), - }) + if (Object.keys(payload).length > 0) { + await fetchApi(`/api/vms/${selectedVM.vmid}/config`, { + method: "POST", + body: JSON.stringify(payload), + }) + } + if (watchdogChanged) { + const updated = await fetchApi(`/api/lxc/${selectedVM.vmid}/oci-watchdog`, { + method: "POST", + body: JSON.stringify({ enabled: pendingWatchdog }), + }) + setOciInstance((prev) => (prev ? { ...prev, watchdog: !!updated?.watchdog } : prev)) + } // Optimistic local reflect so the UI doesn't wait a poll cycle. if (payload.onboot !== undefined) { setVMDetails((prev) => (prev ? { ...prev, config: { ...prev.config, onboot: payload.onboot as number } } : prev)) @@ -1725,6 +1740,7 @@ export function VirtualMachines() { // list card too without waiting up to 2.5 s. void mutate() setPendingOnboot(null) + setPendingWatchdog(null) setPendingTags(null) setNewTagDraft("") setEditingTagIndex(null) @@ -4091,6 +4107,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => { savingOnboot || ( (pendingOnboot === null || pendingOnboot === !!vmDetails.config.onboot) && + (pendingWatchdog === null || !ociInstance?.oci_instance || pendingWatchdog === !!ociInstance.watchdog) && (pendingTags === null || stringifyTags(pendingTags) === stringifyTags(parseTags(selectedVM?.tags))) ) } @@ -4323,6 +4340,34 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => { /> + {/* Watchdog of an OCI application: started + again when it stops on its own. */} + {ociInstance?.oci_instance && ( +
+
+ +
+
+ {t("vmLxc.details.watchdog")} +
+
+ {t("vmLxc.details.watchdogHelp")} +
+
+
+ setPendingWatchdog(v)} + className={`data-[state=checked]:bg-blue-600 data-[state=unchecked]:bg-input border border-border ${!resourcesEditMode ? "opacity-60" : ""}`} + /> +
+ )} + {/* IP Addresses with proper keys */} {selectedVM?.type === "lxc" && vmDetails?.lxc_ip_info && (
diff --git a/AppImage/messages/de/common.json b/AppImage/messages/de/common.json index f7c80dd6..5a6ad367 100644 --- a/AppImage/messages/de/common.json +++ b/AppImage/messages/de/common.json @@ -1049,6 +1049,8 @@ "mounted": "montiert" }, "startOnBoot": "Beim Booten beginnen", + "watchdog": "Watchdog", + "watchdogHelp": "Startet diese Anwendung automatisch neu, wenn sie abstürzt.", "tags": "Tags", "tagsPlaceholder": "Tag hinzufügen…", "tagsNone": "Keine Tags", @@ -2008,6 +2010,8 @@ "oci_recreate_failed": "Neuerstellung der OCI-Anwendung fehlgeschlagen", "oci_modify_completed": "OCI-Anwendung geändert", "oci_modify_failed": "Änderung der OCI-Anwendung fehlgeschlagen", + "oci_watchdog_restarted": "OCI-Anwendung vom Watchdog neu gestartet", + "oci_watchdog_failed": "OCI-Anwendung stoppt immer wieder", "app_update_available": "App-Update verfügbar", "lxc_update_applied": "LXC Update angewendet", "docker_stack_update_available": "Docker Updates verfügbar" @@ -6441,6 +6445,16 @@ "body": "Die Änderung von {app_name} wurde nicht abgeschlossen.\nGrund: {reason}\nContainer: {containers}\nÖffnen Sie „Installierte OCI-Anwendungen verwalten“, um den Zustand zu prüfen.", "label": "Änderung der OCI-Anwendung fehlgeschlagen" }, + "oci_watchdog_restarted": { + "title": "{hostname}: {app_name} nach einem Stopp neu gestartet", + "body": "{app_name} wurde von selbst beendet und vom Watchdog neu gestartet.\nContainer: {containers}", + "label": "OCI-Anwendung vom Watchdog neu gestartet" + }, + "oci_watchdog_failed": { + "title": "{hostname}: {app_name} stoppt immer wieder", + "body": "{app_name} stoppt immer wieder von selbst und wird vom Watchdog nicht mehr neu gestartet.\nContainer: {containers}\nStarten Sie die Anwendung von Hand, sobald die Ursache behoben ist; der Watchdog überwacht sie dann wieder.", + "label": "OCI-Anwendung stoppt immer wieder" + }, "app_update_available": { "title": "{hostname}: {app_name} Update verfügbar auf CT {vmid}", "body": "{app_name} auf CT {vmid} ({ct_name}) hat eine neue Version:\n {installed} → {latest}", diff --git a/AppImage/messages/en/common.json b/AppImage/messages/en/common.json index 6deb9c58..3b0f30f3 100644 --- a/AppImage/messages/en/common.json +++ b/AppImage/messages/en/common.json @@ -1073,6 +1073,8 @@ "configuredButNotMounted": "Configured but not mounted" }, "startOnBoot": "Start at boot", + "watchdog": "Watchdog", + "watchdogHelp": "Automatically restarts this application when it crashes.", "tags": "Tags", "tagsPlaceholder": "Add tag…", "tagsNone": "No tags" @@ -1996,6 +1998,8 @@ "oci_recreate_failed": "OCI application recreation failed", "oci_modify_completed": "OCI application modified", "oci_modify_failed": "OCI application modification failed", + "oci_watchdog_restarted": "OCI application restarted by the watchdog", + "oci_watchdog_failed": "OCI application keeps stopping", "app_update_available": "App update available", "ai_model_migrated": "AI model updated automatically", "proxmenux_update": "ProxMenux update available", @@ -6372,5 +6376,5 @@ "cancel": "Cancel" } }, - "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"The {category} issue has been resolved.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Duration: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"oci_update_completed":{"title":"{hostname}: {app_name} updated","body":"{app_name} was updated to its new image and its data was kept.\nContainers: {containers}","label":"OCI application updated"},"oci_update_failed":{"title":"{hostname}: {app_name} update did not complete","body":"The update of {app_name} did not complete.\nReason: {reason}\nContainers: {containers}\nOpen Manage installed OCI applications to check its state.","label":"OCI application update failed"},"oci_recreate_completed":{"title":"{hostname}: {app_name} recreated","body":"{app_name} was recreated with its saved configuration and its data was kept.\nContainers: {containers}","label":"OCI application recreated"},"oci_recreate_failed":{"title":"{hostname}: {app_name} recreation did not complete","body":"The recreation of {app_name} did not complete.\nReason: {reason}\nContainers: {containers}\nOpen Manage installed OCI applications to check its state.","label":"OCI application recreation failed"},"oci_modify_completed":{"title":"{hostname}: {app_name} modified","body":"{app_name} was modified with its new options and its data was kept.\nContainers: {containers}","label":"OCI application modified"},"oci_modify_failed":{"title":"{hostname}: {app_name} modification did not complete","body":"The modification of {app_name} did not complete.\nReason: {reason}\nContainers: {containers}\nOpen Manage installed OCI applications to check its state.","label":"OCI application modification failed"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname} → {storage}: Backup complete — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}\nThe node is now fully ready to use.","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or open the ProxMenux menu → Settings post-install Proxmox → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Security > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed","completed_with_warnings":"Completed with warnings"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"backup":{"unconfirmedTitle":"{hostname}: Backup outcome unconfirmed","confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed.","warningTitle":"{hostname}: Backup completed with warnings","warningBody":"Backup completed with warnings.","diagnosticsOmitted":"Additional diagnostic lines or text omitted: {count}. Original report retained."}}} + "runtime": {"notifications":{"templates":{"state_change":{"title":"{hostname}: {category} changed to {current}{entity_suffix}","body":"{category} status changed from {previous} to {current}.\n{reason}","label":"Health state changed"},"new_error":{"title":"{hostname}: New {severity} - {category}{entity_suffix}","body":"{reason}","label":"New health issue"},"error_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"The {category} issue has been resolved.\n{reason}\n🚦 Previous severity: {original_severity}\n⏱️ Duration: {duration}","label":"Recovery notification"},"error_escalated":{"title":"{hostname}: Escalated to {severity} - {category}{entity_suffix}","body":"{reason}","label":"Health issue escalated"},"health_degraded":{"title":"{title_or_default}","body":"{reason}","label":"Health check degraded"},"lxc_updates_available":{"title":"{hostname}: {count} LXC(s) with package updates available","body":"📊 {count} LXC(s) with pending package updates (📦 {total_packages} total, 🔒 {security_count} security):\n\n{ct_list}","label":"LXC updates available (experimental)"},"lxc_update_applied":{"title":"{hostname}: LXC {ct_name} ({vmid}) update {result}","body":"{details}","label":"LXC update applied"},"oci_update_completed":{"title":"{hostname}: {app_name} updated","body":"{app_name} was updated to its new image and its data was kept.\nContainers: {containers}","label":"OCI application updated"},"oci_update_failed":{"title":"{hostname}: {app_name} update did not complete","body":"The update of {app_name} did not complete.\nReason: {reason}\nContainers: {containers}\nOpen Manage installed OCI applications to check its state.","label":"OCI application update failed"},"oci_recreate_completed":{"title":"{hostname}: {app_name} recreated","body":"{app_name} was recreated with its saved configuration and its data was kept.\nContainers: {containers}","label":"OCI application recreated"},"oci_recreate_failed":{"title":"{hostname}: {app_name} recreation did not complete","body":"The recreation of {app_name} did not complete.\nReason: {reason}\nContainers: {containers}\nOpen Manage installed OCI applications to check its state.","label":"OCI application recreation failed"},"oci_modify_completed":{"title":"{hostname}: {app_name} modified","body":"{app_name} was modified with its new options and its data was kept.\nContainers: {containers}","label":"OCI application modified"},"oci_modify_failed":{"title":"{hostname}: {app_name} modification did not complete","body":"The modification of {app_name} did not complete.\nReason: {reason}\nContainers: {containers}\nOpen Manage installed OCI applications to check its state.","label":"OCI application modification failed"},"oci_watchdog_restarted":{"title":"{hostname}: {app_name} restarted after it stopped","body":"{app_name} stopped on its own and the watchdog started it again.\nContainers: {containers}","label":"OCI application restarted by the watchdog"},"oci_watchdog_failed":{"title":"{hostname}: {app_name} keeps stopping","body":"{app_name} keeps stopping on its own and the watchdog no longer restarts it.\nContainers: {containers}\nStart it by hand once the cause is solved; the watchdog then watches it again.","label":"OCI application keeps stopping"},"app_update_available":{"title":"{hostname}: {app_name} update available on CT {vmid}","body":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","label":"App update available"},"docker_stack_update_available":{"title":"{hostname}: Docker updates available on CT {vmid}","body":"Container {ct_name} (CT {vmid}) has {count} Docker update(s):\n{details}","label":"Docker updates available"},"vm_start":{"title":"{hostname}: VM {vmname} ({vmid}) started","body":"Virtual machine {vmname} (ID: {vmid}) is now running.","label":"VM started"},"vm_start_warning":{"title":"{hostname}: VM {vmname} ({vmid}) started with warnings","body":"Virtual machine {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"VM started (warnings)"},"vm_stop":{"title":"{hostname}: VM {vmname} ({vmid}) stopped","body":"Virtual machine {vmname} (ID: {vmid}) has been stopped.","label":"VM stopped"},"vm_shutdown":{"title":"{hostname}: VM {vmname} ({vmid}) shut down","body":"Virtual machine {vmname} (ID: {vmid}) has been cleanly shut down.","label":"VM shutdown"},"vm_fail":{"title":"{hostname}: VM {vmname} ({vmid}) FAILED","body":"Virtual machine {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"VM FAILED"},"vm_restart":{"title":"{hostname}: VM {vmname} ({vmid}) restarted","body":"Virtual machine {vmname} (ID: {vmid}) has been restarted.","label":"VM restarted"},"ct_start":{"title":"{hostname}: CT {vmname} ({vmid}) started","body":"Container {vmname} (ID: {vmid}) is now running.","label":"CT started"},"ct_start_warning":{"title":"{hostname}: CT {vmname} ({vmid}) started with warnings","body":"Container {vmname} (ID: {vmid}) started successfully but has warnings.\nWarnings: {reason}","label":"CT started (warnings)"},"ct_stop":{"title":"{hostname}: CT {vmname} ({vmid}) stopped","body":"Container {vmname} (ID: {vmid}) has been stopped.","label":"CT stopped"},"ct_shutdown":{"title":"{hostname}: CT {vmname} ({vmid}) shut down","body":"Container {vmname} (ID: {vmid}) has been cleanly shut down.","label":"CT shutdown"},"ct_restart":{"title":"{hostname}: CT {vmname} ({vmid}) restarted","body":"Container {vmname} (ID: {vmid}) has been restarted.","label":"CT restarted"},"ct_fail":{"title":"{hostname}: CT {vmname} ({vmid}) FAILED","body":"Container {vmname} (ID: {vmid}) has crashed or failed to start.\nReason: {reason}","label":"CT FAILED"},"migration_start":{"title":"{hostname}: Migration started — {vmname} ({vmid})","body":"Live migration of {vmname} (ID: {vmid}) to node {target_node} has started.","label":"Migration started"},"migration_complete":{"title":"{hostname}: Migration complete — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) successfully migrated to node {target_node}.","label":"Migration complete"},"migration_warning":{"title":"{hostname}: Migration complete with warnings — {vmname} ({vmid})","body":"{vmname} (ID: {vmid}) migrated to node {target_node} but encountered warnings.\nWarnings: {reason}","label":"Migration (warnings)"},"migration_fail":{"title":"{hostname}: Migration FAILED — {vmname} ({vmid})","body":"Migration of {vmname} (ID: {vmid}) to node {target_node} failed.\nReason: {reason}","label":"Migration FAILED"},"replication_fail":{"title":"{hostname}: Replication FAILED — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Replication FAILED"},"replication_complete":{"title":"{hostname}: Replication complete — {vmname} ({vmid})","body":"Replication of {vmname} (ID: {vmid}) completed successfully.","label":"Replication complete"},"backup_start":{"title":"{hostname}: Backup started on {storage}","body":"Backup job started on storage {storage}.\n{reason}","label":"Backup started"},"backup_complete":{"title":"{hostname} → {storage}: Backup complete — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}","label":"Backup complete"},"backup_warning":{"title":"{hostname} → {storage}: Backup complete with warnings — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}","label":"Backup (warnings)"},"backup_fail":{"title":"{hostname} → {storage}: Backup FAILED — {vmname} ({vmid})","body":"Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}","label":"Backup FAILED"},"host_backup_start":{"title":"{hostname}: Host backup started → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}","label":"Host backup started"},"host_backup_complete":{"title":"{hostname}: Host backup complete → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}","label":"Host backup complete"},"host_backup_fail":{"title":"{hostname}: Host backup FAILED → {backend_label}","body":"Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}","label":"Host backup FAILED"},"snapshot_complete":{"title":"{hostname}: Snapshot created — {vmname} ({vmid})","body":"Snapshot \"{snapshot_name}\" created for {vmname} (ID: {vmid}).","label":"Snapshot created"},"snapshot_fail":{"title":"{hostname}: Snapshot FAILED — {vmname} ({vmid})","body":"Snapshot creation for {vmname} (ID: {vmid}) failed.\nReason: {reason}","label":"Snapshot FAILED"},"cpu_high":{"title":"{hostname}: High CPU usage — {value}%","body":"CPU usage has reached {value}% on {cores} cores.\n{details}","label":"High CPU usage"},"ram_high":{"title":"{hostname}: High memory usage — {value}%","body":"Memory usage: {used} / {total} ({value}%).\n{details}","label":"High memory usage"},"temp_high":{"title":"{hostname}: High sensor temperature — {value}°C","body":"Sensor temperature has reached {value}°C (threshold: {threshold}°C).\n{details}","label":"High temperature"},"disk_space_low":{"title":"{hostname}: Low disk space on {mount}","body":"Filesystem {mount}: {used}% used ({available} available).\nFree up disk space to avoid service disruption.","label":"Low disk space"},"disk_io_error":{"title":"{hostname}: Disk failure detected — {device}","body":"I/O error or disk failure detected on device {device}.\n{reason}","label":"Disk failure / I/O error"},"storage_unavailable":{"title":"{hostname}: Storage unavailable — {storage_name}","body":"PVE storage \"{storage_name}\" (type: {storage_type}) is not accessible.\nReason: {reason}","label":"Storage unavailable"},"smart_test_complete":{"title":"{hostname}: SMART test completed — {device}","body":"SMART {test_type} test on /dev/{device} has completed.\nResult: {result}\nDuration: {duration}","label":"SMART test completed"},"smart_test_failed":{"title":"{hostname}: SMART test FAILED — {device}","body":"SMART {test_type} test on /dev/{device} has failed.\nResult: {result}\nReason: {reason}","label":"SMART test FAILED"},"gpu_mode_switch":{"title":"{hostname}: GPU mode changed to {new_mode}","body":"GPU passthrough mode has been switched.\nGPU: {gpu_name} ({gpu_pci})\nPrevious mode: {old_mode}\nNew mode: {new_mode}\n{details}","label":"GPU mode switched"},"gpu_passthrough_blocked":{"title":"{hostname}: {guest_type} {guest_id} blocked at startup","body":"PCIe passthrough guard prevented {guest_type} {guest_id} ({guest_name}) from starting.\nReason: {reason}\n{details}","label":"GPU passthrough blocked"},"pci_passthrough_conflict":{"title":"{hostname}: PCIe device conflict detected — {device_pci}","body":"A PCIe device is assigned to multiple guests.\nDevice: {device_pci}\nConflicting guests: {guest_list}\nAction required: Stop one of the guests or reassign the device.","label":"PCIe device conflict"},"load_high":{"title":"{hostname}: High system load — {value}","body":"System load average is {value} on {cores} cores.\n{details}","label":"High system load"},"network_down":{"title":"{hostname}: Network connectivity lost{entity_suffix}","body":"The node has lost network connectivity.\nReason: {reason}","label":"Network connectivity lost"},"network_latency":{"title":"{hostname}: High network latency — {value}ms","body":"Latency to gateway: {value}ms (threshold: {threshold}ms).\nThis may indicate network congestion or hardware issues.","label":"High network latency"},"auth_fail":{"title":"{hostname}: Authentication failure","body":"Failed login attempt detected.\nSource IP: {source_ip}\nUser: {username}\nService: {service}","label":"Authentication failure"},"ip_block":{"title":"{hostname}: IP blocked by Fail2Ban","body":"IP address {source_ip} has been banned.\nJail: {jail}\nFailed attempts: {failures}","label":"IP blocked by Fail2Ban"},"firewall_issue":{"title":"{hostname}: Firewall issue detected{entity_suffix}","body":"A firewall configuration issue has been detected.\nReason: {reason}","label":"Firewall issue detected"},"user_permission_change":{"title":"{hostname}: User permission changed","body":"User: {username}\nChange: {change_details}","label":"User permission changed"},"split_brain":{"title":"{hostname}: Cluster event reported","body":"A cluster event was reported:\n{reason}","label":"Cluster event"},"node_disconnect":{"title":"{hostname}: Node {node_name} disconnected","body":"Node {node_name} has disconnected from the cluster.","label":"Node disconnected"},"node_reconnect":{"title":"{hostname}: Node {node_name} reconnected","body":"Node {node_name} has rejoined the cluster successfully.","label":"Node reconnected"},"system_startup":{"title":"{hostname}: {reason}","body":"{summary}","label":"System startup report"},"system_shutdown":{"title":"{hostname}: System shutting down","body":"The node is shutting down.\n{reason}","label":"System shutting down"},"system_reboot":{"title":"{hostname}: System rebooting","body":"The node is rebooting.\n{reason}","label":"System rebooting"},"system_restore_completed":{"title":"{hostname}: Host restore finished","body":"Post-restore tasks completed in background.\n\nGuests applied: {guests}\nBind-mount stubs: {stubs}\nStale node dirs removed: {stale_nodes}\nComponents reinstalled: {components}\nDuration: {duration}\n{warnings_block}\nThe node is now fully ready to use.","label":"Host restore completed"},"system_problem":{"title":"{hostname}: System problem detected{entity_suffix}","body":"A system-level problem has been detected.\nReason: {reason}","label":"System problem detected"},"service_fail":{"title":"{hostname}: Service failed — {service_name}","body":"System service \"{service_name}\" has failed.\nReason: {reason}","label":"Service failed"},"oom_kill":{"title":"{hostname}: OOM Kill — {process}","body":"Process \"{process}\" was killed by the Out-of-Memory manager.\n{reason}","label":"Out of memory kill"},"service_fail_batch":{"title":"{hostname}: {service_count} services failed{entity_suffix}","body":"{reason}","label":"Service fail batch"},"cron_output":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Cron job output (per-cron stdout via mail)"},"system_mail":{"title":"{hostname}: {pve_title}","body":"{reason}","label":"Smartd / mail bounces (PVE system mail)"},"apt_listchanges":{"title":"{hostname}: {pve_title}","body":"Upstream package information forwarded by Proxmox VE through apt-listchanges. The following text comes from the package maintainer and is not a ProxMenux recommendation.\n\n{reason}","label":"apt-listchanges package notices"},"webhook_test":{"title":"{hostname}: Webhook test received","body":"PVE webhook connectivity test successful.\n{reason}","label":"Webhook test"},"update_available":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity: {security_count}\nProxmox: {pve_count}\nKernel: {kernel_count}\nImportant packages:\n{important_list}","label":"Updates available (legacy)"},"unknown_persistent":{"title":"{hostname}: Check unavailable - {category}{entity_suffix}","body":"Health check for {category} has been unavailable for 3+ cycles.\n{reason}","label":"Check unavailable"},"health_persistent":{"title":"{hostname}: {count} active health issue(s){entity_suffix}","body":"The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.","label":"Active health issues (daily)"},"health_issue_new":{"title":"{hostname}: New health issue — {category}{entity_suffix}","body":"New {severity} issue detected in: {category}\nDetails: {reason}","label":"New health issue"},"health_issue_resolved":{"title":"{hostname}: Resolved - {category}{entity_suffix}","body":"{category} issue has been resolved.\n{reason}\nDuration: {duration}","label":"Health issue resolved"},"update_summary":{"title":"{hostname}: Updates available","body":"Total updates: {total_count}\nSecurity updates: {security_count}\nProxmox-related updates: {pve_count}\nKernel updates: {kernel_count}\nImportant packages:\n{important_list}","label":"Host package updates"},"pve_update":{"title":"{hostname}: Proxmox VE {new_version} available","body":"A new Proxmox VE release is available.\nCurrent: {current_version}\nNew: {new_version}\n{details}","label":"Proxmox VE update available"},"update_complete":{"title":"{hostname}: System update completed","body":"System packages have been successfully updated.\n{details}","label":"Host update completed"},"ai_model_migrated":{"title":"{hostname}: AI model updated","body":"The AI model for notifications has been automatically updated.\nProvider: {provider}\nPrevious model: {old_model}\nNew model: {new_model}\n\n{message}","label":"AI model auto-updated"},"proxmenux_update":{"title":"{hostname}: ProxMenux {new_version} available","body":"A new version of ProxMenux is available.\nCurrent: {current_version}\nNew: {new_version}","label":"ProxMenux update available"},"mount_stale":{"title":"{hostname}: stale remote mount {mount_target}","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is stale{lxc_scope}.\nStat timed out or returned an error: {error}\n\nApps writing to this path will silently land on the underlying filesystem and may fill the disk. Remount or fix connectivity ASAP.","label":"Remote mount stale"},"mount_readonly":{"title":"{hostname}: remote mount {mount_target} is read-only","body":"Remote mount {mount_target} ({fstype}) from {mount_source} is mounted read-only{lxc_scope}. Writes will fail. If this was unintentional, remount with rw.","label":"Remote mount read-only"},"lxc_disk_low":{"title":"{hostname}: CT {vmid} rootfs at {usage_percent}%","body":"CT {vmid} ({name}) rootfs is at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nA full LXC rootfs prevents the container from booting cleanly. Either expand the rootfs (pct resize {vmid} rootfs +1G) or free space inside the container.","label":"LXC rootfs near full"},"vm_disk_low":{"title":"{hostname}: VM {vmid} filesystems at {usage_percent}%","body":"VM {vmid} ({name}) guest filesystems are at {usage_percent}% ({disk_bytes_human} / {maxdisk_bytes_human}).\n\nReported by the QEMU guest agent. Includes every persistent filesystem the guest mounts on a block device — virtual disks and PCI-passthrough drives alike. Free up space inside the guest or expand the affected storage before writes start to fail.","label":"VM filesystems near full"},"lxc_mount_low":{"title":"{hostname}: CT {vmid} mount {mount} at {usage_percent}%","body":"Mount {mount} inside CT {vmid} ({name}) is at {usage_percent}% used.\nFilesystem type: {fstype}\n\nA full mount inside a container often blocks the application silently — writes either fail or, worse, land on the rootfs and trigger the rootfs alert next. Free up space on the mount or expand it.","label":"LXC mount near full"},"pve_storage_full":{"title":"{hostname}: PVE storage {storage_name} at {usage_percent}%","body":"Proxmox storage \"{storage_name}\" (type: {storage_type}) is at {usage_percent}% used.\n\nOnce full, no new VM/CT can be provisioned and existing guests may fail to write. Move/delete unused volumes or expand the underlying pool/LV/RBD image.","label":"PVE storage near full"},"zfs_pool_full":{"title":"{hostname}: ZFS pool {pool_name} at {usage_percent}%","body":"ZFS pool \"{pool_name}\" is at {usage_percent}% capacity.\n\nZFS performance and write reliability degrade sharply above ~80% capacity (CoW needs free space for new blocks). Free up snapshots, prune old datasets, or add more vdevs to the pool.","label":"ZFS pool near full"},"post_install_update":{"title":"{hostname}: {count} ProxMenux optimization update(s) available","body":"{count} ProxMenux optimization update(s) available on this host.\n\n🛠️ Available versions:\n{tool_list}\n\n💡 Apply from:\n • ProxMenux Monitor → Settings → ProxMenux Optimizations\n • Or open the ProxMenux menu → Settings post-install Proxmox → \"Apply available updates\"","label":"ProxMenux optimization updates available"},"secure_gateway_update_available":{"title":"{hostname}: {app_name} update available{update_title_suffix}","body":"{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) pending in its container.\n{version_line}\n\n💡 Open ProxMenux Monitor > Security > Secure Gateway and click \"Update\" to apply.\n\n🗂️ Packages:\n{package_list}","label":"Secure Gateway update available"},"nvidia_driver_update_available":{"title":"{hostname}: NVIDIA driver update available — v{latest_version}","body":"A newer maintenance release is available for the installed NVIDIA driver branch.\n🔹 Currently installed: v{current_version}\n🟢 Latest available: v{latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\nReinstalling rebuilds the DKMS module against the running kernel and requires a reboot to load the new driver.","label":"NVIDIA driver update available"},"coral_driver_update_available":{"title":"{hostname}: Coral TPU driver update available — {latest_version}","body":"A newer {variant_label} is available.\n🔹 Currently installed: {current_version}\n🟢 Latest available: {latest_version}\n\n{upgrade_reason}\n\n💡 To reinstall:\n • From the ProxMenux post-install menu: {menu_label}\n\n{reboot_note}","label":"Coral TPU driver update available"},"burst_auth_fail":{"title":"{hostname}: +{count} more auth failures in {window}","body":"+{count} additional authentication failures detected in {window} ({total_count} total).\nSources: {entity_list}","label":"Auth failures burst"},"burst_ip_block":{"title":"{hostname}: Fail2Ban banned +{count} more IPs in {window}","body":"+{count} additional IPs banned by Fail2Ban in {window} ({total_count} total).\nIPs: {entity_list}","label":"IP block burst"},"burst_disk_io":{"title":"{hostname}: +{count} more disk I/O errors on {entity_list}","body":"+{count} additional I/O errors detected in {window} ({total_count} total).\nDevices: {entity_list}","label":"Disk I/O burst"},"burst_cluster":{"title":"{hostname}: Cluster flapping detected (+{count} more changes)","body":"Cluster state changed +{count} more times in {window} ({total_count} total).\nNodes: {entity_list}","label":"Cluster flapping burst"},"burst_service_fail":{"title":"{hostname}: +{count} more services failed in {window}","body":"+{count} additional service failures detected in {window} ({total_count} total).\nThis typically indicates a node reboot or PVE service restart.\n\nAdditional failures:\n{details}","label":"Service fail burst"},"burst_system":{"title":"{hostname}: +{count} more system problems in {window}","body":"+{count} additional system problems detected in {window} ({total_count} total).\n\nAdditional issues:\n{details}","label":"System problems burst"},"burst_generic":{"title":"{hostname}: +{count} more {event_type} events in {window}","body":"+{count} additional events of type {event_type} in {window} ({total_count} total).\n\nAdditional events:\n{details}","label":"Generic burst"},"kernel_warning":{"title":"{hostname}: Kernel diagnostic event detected","body":"The kernel recorded a diagnostic event: {kernel_details}","label":"Kernel warnings and diagnostic traces"}},"lxcUpdate":{"status":{"success":"succeeded","failure":"failed","partial":"completed partially","deferred":"deferred","skipped":"skipped"},"source":{"scheduled":"Scheduled","manual":"Manual"},"sourceLabel":"Source","targets":"Targets","osPending":"OS packages pending: {before} → {after}","osUnverified":"OS packages: update command executed; result not verified","applications":"Applications: {items}","applicationsUnverified":"Applications: updater executed; no tracked version change was observed","dockerEngineChange":"Docker Engine: {before} → {after}","dockerEngineVerified":"Docker Engine: verified at {version}","dockerEngineUnverified":"Docker Engine: update command executed; version not verified","dockerImagesPending":"Docker images pending: {before} → {after}","dockerImagesChanged":"Docker images changed: {images}","deferredTargets":"Deferred targets: {targets}","reason":"Reason: {reason}","restartRequired":"Restart required: {value}","yes":"yes","no":"no","verificationPending":"Verification pending until the container is running","verificationWarning":"Verification warning: {error}","duration":"Duration: {duration}"},"fallback":{"unknownTitle":"{hostname}: {event_type}","healthCheckDegraded":"Health check degraded","none":"none","temperatureAlertTitle":"{hostname}: Sensor temperature alert","temperatureAlertBody":"Temperature alert recorded without a complete measurement payload.","recordedReason":"Recorded reason: {reason}","recordedDetails":"Recorded details: {details}"},"fields":{"vmid":"VM/CT","name":"Name","device":"Device","sourceIp":"Source IP","node":"Node","category":"Category","service":"Service","jail":"Jail","user":"User","count":"Count","window":"Window","affected":"Affected"},"vzdump":{"size":"Size: {value}","duration":"Duration: {value}","file":"File","backups":"{count} backups","failed":"{count} failed","total":"Total: {value}","time":"Time: {value}"},"startup":{"issuesTitle":"{hostname}: System startup - {count} issue(s) detected","completeTitle":"{hostname}: System startup completed","operational":"All systems operational.","vmCountOne":"{count} VM","vmCountMany":"{count} VMs","ctCountOne":"{count} CT","ctCountMany":"{count} CTs","started":"✅ {counts} started","vmFailed":"❌ VM failed: {name} - {reason}","ctFailed":"❌ CT failed: {name} - {reason}","unknownError":"unknown error","storageUnavailable":"⚠️ Storage: {count} unavailable ({names})","servicesFailed":"⚠️ Services: {count} failed ({names})","duration":"⏱️ Startup completed in {minutes} min"},"appUpdates":{"singleTitle":"{hostname}: {app_name} update available on CT {vmid}","singleBody":"{app_name} on CT {vmid} ({ct_name}) has a new version:\n {installed} → {latest}","emptyTitle":"{hostname}: Application updates available","emptyBody":"Application updates are available.","batchTitle":"{hostname}: {count} application updates available","batchLeadOneContainer":"{count} applications in {container_count} LXC container have a newer version:","batchLeadManyContainers":"{count} applications in {container_count} LXC containers have a newer version:","additional":"… {count} additional application(s)","app":"app","unknown":"unknown"},"healthDegraded":{"categories":{"cpu":"CPU usage and temperature","memory":"Memory and swap","storage":"Storage mounts and space","disks":"Disk I/O and errors","network":"Network interfaces","vms":"VMs and containers","services":"PVE services","logs":"System logs","updates":"System updates","security":"Security"},"severity":{"critical":"Critical","warning":"Warning"},"singleTitle":"{hostname}: {severity} health status – {category}","multipleTitle":"{hostname}: {count} health checks degraded","multipleLine":"• {severity}: {category}\n {reason}","reasons":{"cpuSustained":"CPU is above {threshold}% for {duration}s."}},"digest":{"title":"{hostname}: 24h summary ({timestamp})","readFailed":"digest read failed: {error}","empty":"No INFO events buffered for this digest window.","lead":"{count} INFO events grouped by category:\n","more":" • … and {count} more","footer":"(Critical/Warning events arrived at the time they happened, not in this digest.)","quietTitle":"{hostname}: {count} events buffered during Quiet Hours","groups":{"vm_ct":"VM a CT","backup":"Backup","resources":"Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"Services","health":"Health","updates":"Updates","hardware":"Hardware","system":"System","other":"Other"}},"test":{"title":"ProxMenux Test","welcome":"Welcome to ProxMenux Monitor!","verify":"This is a test message to verify your notification channel is working correctly.","configuration":"Channel configuration:","iconsEnabled":"Icons: enabled","iconsDisabled":"Icons: disabled","aiEnabled":"AI: enabled ({info})","aiDisabled":"AI: disabled","alerts":"You will receive alerts about system events, health status changes, and security incidents.","photoCaption":"You can use this image as the profile photo for your notification bot."},"channels":{"discord":{"category":"Category","host":"Host","severity":"Severity"},"email":{"report":"{group} Report","details":"Details","host":"Host","footer":"ProxMenux Notification Service","severity":{"critical":"Critical","warning":"Warning","info":"Information","ok":"Resolved","default":"Notice"},"groups":{"vm_ct":"Virtual Machine / Container","backup":"Backup & Snapshot","resources":"System Resources","storage":"Storage","network":"Network","security":"Security","cluster":"Cluster","services":"System Services","health":"Health Monitor","updates":"System Updates","system":"System","hardware":"Hardware","other":"System Notification"},"fields":{"vmCtId":"VM/CT ID","name":"Name","action":"Action","targetNode":"Target Node","reason":"Reason","storage":"Storage","status":"Status","size":"Size","duration":"Duration","snapshot":"Snapshot","metric":"Metric","currentValue":"Current Value","threshold":"Threshold","cpuCores":"CPU Cores","memory":"Memory","temperature":"Temperature","mountPoint":"Mount Point","usage":"Usage","available":"Available","device":"Device","severity":"Severity","storageName":"Storage Name","type":"Type","interface":"Interface","latency":"Latency","event":"Event","sourceIp":"Source IP","username":"Username","service":"Service","jail":"Jail","failures":"Failures","change":"Change","node":"Node","quorum":"Quorum","nodesAffected":"Nodes Affected","process":"Process","category":"Category","previousSeverity":"Previous Severity","activeIssues":"Active Issues","totalUpdates":"Total Updates","securityUpdates":"Security Updates","proxmoxUpdates":"Proxmox Updates","kernelUpdates":"Kernel Updates","importantPackages":"Important Packages","currentVersion":"Current Version","newVersion":"New Version"},"status":{"failed":"Failed","completed":"Completed","started":"Started","unconfirmed":"Unconfirmed","completed_with_warnings":"Completed with warnings"}}},"temperature":{"sampleSpan":"High samples span {duration}."},"backup":{"unconfirmedTitle":"{hostname}: Backup outcome unconfirmed","confirmedTitle":"{hostname}: Backup complete","confirmedBody":"Backup completed successfully.","errorTitle":"{hostname}: Backup error reported","errorBody":"The backup report contains an error.","unconfirmedBody":"The backup outcome is not confirmed.","warningTitle":"{hostname}: Backup completed with warnings","warningBody":"Backup completed with warnings.","diagnosticsOmitted":"Additional diagnostic lines or text omitted: {count}. Original report retained."}}} } diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index f2257a32..e8b50751 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -1049,6 +1049,8 @@ "mounted": "montado" }, "startOnBoot": "Iniciar al arrancar", + "watchdog": "Vigilancia", + "watchdogHelp": "Reinicia automáticamente esta aplicación cuando falla.", "tags": "Etiquetas", "tagsPlaceholder": "Añadir etiqueta…", "tagsNone": "Sin etiquetas", @@ -1995,6 +1997,8 @@ "oci_recreate_failed": "Recreación de aplicación OCI fallida", "oci_modify_completed": "Aplicación OCI modificada", "oci_modify_failed": "Modificación de aplicación OCI fallida", + "oci_watchdog_restarted": "Aplicación OCI reiniciada por la vigilancia", + "oci_watchdog_failed": "Aplicación OCI que se detiene una y otra vez", "app_update_available": "Actualización de app disponible", "ai_model_migrated": "Modelo de IA actualizado automáticamente", "proxmenux_update": "Actualización de ProxMenux disponible", @@ -6441,6 +6445,16 @@ "body": "La modificación de {app_name} no se completó.\nMotivo: {reason}\nContenedores: {containers}\nAbre Gestionar aplicaciones OCI instaladas para comprobar su estado.", "label": "Modificación de aplicación OCI fallida" }, + "oci_watchdog_restarted": { + "title": "{hostname}: {app_name} reiniciado tras detenerse", + "body": "{app_name} se detuvo solo y la vigilancia lo ha iniciado de nuevo.\nContenedores: {containers}", + "label": "Aplicación OCI reiniciada por la vigilancia" + }, + "oci_watchdog_failed": { + "title": "{hostname}: {app_name} se detiene una y otra vez", + "body": "{app_name} sigue deteniéndose solo y la vigilancia ya no lo reinicia.\nContenedores: {containers}\nInícialo a mano cuando la causa esté resuelta; la vigilancia volverá a vigilarlo.", + "label": "Aplicación OCI que se detiene una y otra vez" + }, "app_update_available": { "title": "{hostname}: actualización de {app_name} disponible en CT {vmid}", "body": "{app_name} en CT {vmid} ({ct_name}) tiene una nueva versión:\n {installed} → {latest}", diff --git a/AppImage/messages/fr/common.json b/AppImage/messages/fr/common.json index 68b496af..34c3cb82 100644 --- a/AppImage/messages/fr/common.json +++ b/AppImage/messages/fr/common.json @@ -1049,6 +1049,8 @@ "mounted": "monté" }, "startOnBoot": "Démarrer au démarrage", + "watchdog": "Surveillance", + "watchdogHelp": "Redémarre automatiquement cette application lorsqu'elle plante.", "tags": "balises", "tagsPlaceholder": "Ajouter une balise…", "tagsNone": "Aucune balise", @@ -2008,6 +2010,8 @@ "oci_recreate_failed": "Échec de la recréation de l'application OCI", "oci_modify_completed": "Application OCI modifiée", "oci_modify_failed": "Échec de la modification de l'application OCI", + "oci_watchdog_restarted": "Application OCI redémarrée par la surveillance", + "oci_watchdog_failed": "Application OCI qui s'arrête sans cesse", "app_update_available": "mise à jour de l'application disponible", "lxc_update_applied": "Mise à jour LXC appliquée", "docker_stack_update_available": "Docker mises à jour disponibles" @@ -6441,6 +6445,16 @@ "body": "La modification de {app_name} n'a pas abouti.\nMotif : {reason}\nConteneurs : {containers}\nOuvrez Gérer les applications OCI installées pour vérifier son état.", "label": "Échec de la modification de l'application OCI" }, + "oci_watchdog_restarted": { + "title": "{hostname} : {app_name} redémarré après un arrêt", + "body": "{app_name} s'est arrêté tout seul et la surveillance l'a redémarré.\nConteneurs : {containers}", + "label": "Application OCI redémarrée par la surveillance" + }, + "oci_watchdog_failed": { + "title": "{hostname} : {app_name} s'arrête sans cesse", + "body": "{app_name} continue de s'arrêter tout seul et la surveillance ne le redémarre plus.\nConteneurs : {containers}\nDémarrez-le à la main une fois la cause résolue ; la surveillance le reprend alors.", + "label": "Application OCI qui s'arrête sans cesse" + }, "app_update_available": { "title": "{hostname} : mise à jour {app_name} disponible sur CT {vmid}", "body": "{app_name} sur CT {vmid} ({ct_name}) a une nouvelle version :\n {installed} → {latest}", diff --git a/AppImage/messages/it/common.json b/AppImage/messages/it/common.json index 39e5d561..ea8eb818 100644 --- a/AppImage/messages/it/common.json +++ b/AppImage/messages/it/common.json @@ -1049,6 +1049,8 @@ "mounted": "montato" }, "startOnBoot": "Avvio al boot", + "watchdog": "Watchdog", + "watchdogHelp": "Riavvia automaticamente questa applicazione quando si arresta in modo anomalo.", "tags": "Tag", "tagsPlaceholder": "Aggiungi tag…", "tagsNone": "Nessun tag", @@ -2008,6 +2010,8 @@ "oci_recreate_failed": "Ricreazione app OCI non riuscita", "oci_modify_completed": "App OCI modificata", "oci_modify_failed": "Modifica app OCI non riuscita", + "oci_watchdog_restarted": "App OCI riavviata dal watchdog", + "oci_watchdog_failed": "App OCI che continua ad arrestarsi", "app_update_available": "Aggiornamento app disponibile", "lxc_update_applied": "Aggiornamento LXC applicato", "docker_stack_update_available": "Aggiornamenti Docker disponibili" @@ -6441,6 +6445,16 @@ "body": "La modifica di {app_name} non è stata completata.\nMotivo: {reason}\nContainer: {containers}\nApri «Gestisci le applicazioni OCI installate» per verificarne lo stato.", "label": "Modifica app OCI non riuscita" }, + "oci_watchdog_restarted": { + "title": "{hostname}: {app_name} riavviato dopo un arresto", + "body": "{app_name} si è arrestato da solo e il watchdog lo ha riavviato.\nContainer: {containers}", + "label": "App OCI riavviata dal watchdog" + }, + "oci_watchdog_failed": { + "title": "{hostname}: {app_name} continua ad arrestarsi", + "body": "{app_name} continua ad arrestarsi da solo e il watchdog non lo riavvia più.\nContainer: {containers}\nAvvialo a mano una volta risolta la causa; il watchdog tornerà a sorvegliarlo.", + "label": "App OCI che continua ad arrestarsi" + }, "app_update_available": { "title": "{hostname}: aggiornamento {app_name} disponibile su CT {vmid}", "body": "{app_name} su CT {vmid} ({ct_name}) ha una nuova versione:\n {installed} → {latest}", diff --git a/AppImage/messages/pt/common.json b/AppImage/messages/pt/common.json index d86ebc58..df2e44ab 100644 --- a/AppImage/messages/pt/common.json +++ b/AppImage/messages/pt/common.json @@ -1049,6 +1049,8 @@ "mounted": "montado" }, "startOnBoot": "Iniciar na inicialização", + "watchdog": "Vigilância", + "watchdogHelp": "Reinicia automaticamente esta aplicação quando falha.", "tags": "Etiquetas", "tagsPlaceholder": "Adicionar tag…", "tagsNone": "sem tags", @@ -2008,6 +2010,8 @@ "oci_recreate_failed": "Falha na recriação da aplicação OCI", "oci_modify_completed": "Aplicação OCI modificada", "oci_modify_failed": "Falha na modificação da aplicação OCI", + "oci_watchdog_restarted": "Aplicação OCI reiniciada pela vigilância", + "oci_watchdog_failed": "Aplicação OCI que continua a parar", "app_update_available": "atualização de aplicativo disponível", "lxc_update_applied": "Atualização LXC aplicada", "docker_stack_update_available": "Docker atualizações disponíveis" @@ -6441,6 +6445,16 @@ "body": "A modificação de {app_name} não foi concluída.\nMotivo: {reason}\nContêineres: {containers}\nAbra Gerir aplicações OCI instaladas para verificar o estado.", "label": "Falha na modificação da aplicação OCI" }, + "oci_watchdog_restarted": { + "title": "{hostname}: {app_name} reiniciado depois de parar", + "body": "{app_name} parou sozinho e a vigilância voltou a iniciá-lo.\nContêineres: {containers}", + "label": "Aplicação OCI reiniciada pela vigilância" + }, + "oci_watchdog_failed": { + "title": "{hostname}: {app_name} continua a parar", + "body": "{app_name} continua a parar sozinho e a vigilância já não o reinicia.\nContêineres: {containers}\nInicie-o à mão quando a causa estiver resolvida; a vigilância volta então a vigiá-lo.", + "label": "Aplicação OCI que continua a parar" + }, "app_update_available": { "title": "{hostname}: atualização {app_name} disponível no CT {vmid}", "body": "{app_name} no CT {vmid} ({ct_name}) tem uma nova versão:\n {installed} → {latest}", diff --git a/AppImage/messages/sk/common.json b/AppImage/messages/sk/common.json index fb4b1350..aa35128c 100644 --- a/AppImage/messages/sk/common.json +++ b/AppImage/messages/sk/common.json @@ -1073,6 +1073,8 @@ "configuredButNotMounted": "Nastavené, ale nepripojené" }, "startOnBoot": "Spúšťať pri štarte", + "watchdog": "Watchdog", + "watchdogHelp": "Automaticky reštartuje túto aplikáciu, keď zlyhá.", "tags": "Značky", "tagsPlaceholder": "Pridať značku…", "tagsNone": "Žiadne značky" @@ -2009,6 +2011,8 @@ "oci_recreate_failed": "Opätovné vytvorenie OCI aplikácie zlyhalo", "oci_modify_completed": "OCI aplikácia upravená", "oci_modify_failed": "Úprava OCI aplikácie zlyhala", + "oci_watchdog_restarted": "OCI aplikácia reštartovaná watchdogom", + "oci_watchdog_failed": "OCI aplikácia sa opakovane zastavuje", "app_update_available": "K dispozícii je aktualizácia aplikácie" }, "ui": { @@ -6440,6 +6444,16 @@ "body": "Úprava aplikácie {app_name} sa nedokončila.\nDôvod: {reason}\nKontajnery: {containers}\nOtvorte Spravovať nainštalované OCI aplikácie a skontrolujte jej stav.", "label": "Úprava OCI aplikácie zlyhala" }, + "oci_watchdog_restarted": { + "title": "{hostname}: {app_name} reštartovaná po zastavení", + "body": "Aplikácia {app_name} sa sama zastavila a watchdog ju znova spustil.\nKontajnery: {containers}", + "label": "OCI aplikácia reštartovaná watchdogom" + }, + "oci_watchdog_failed": { + "title": "{hostname}: {app_name} sa opakovane zastavuje", + "body": "Aplikácia {app_name} sa stále sama zastavuje a watchdog ju už nereštartuje.\nKontajnery: {containers}\nPo odstránení príčiny ju spustite ručne; watchdog ju potom bude znova sledovať.", + "label": "OCI aplikácia sa opakovane zastavuje" + }, "app_update_available": { "title": "{hostname}: Pre {app_name} je na CT {vmid} dostupná aktualizácia", "body": "Aplikácia {app_name} na CT {vmid} ({ct_name}) má novú verziu:\n {installed} → {latest}", diff --git a/AppImage/messages/sv/common.json b/AppImage/messages/sv/common.json index 92bcae3a..c95baa74 100644 --- a/AppImage/messages/sv/common.json +++ b/AppImage/messages/sv/common.json @@ -1049,6 +1049,8 @@ "mounted": "monterad" }, "startOnBoot": "Börja vid start", + "watchdog": "Watchdog", + "watchdogHelp": "Startar om den här applikationen automatiskt när den kraschar.", "tags": "Taggar", "tagsPlaceholder": "Lägg till tagg...", "tagsNone": "Inga taggar", @@ -2008,6 +2010,8 @@ "oci_recreate_failed": "Återskapande av OCI-applikation misslyckades", "oci_modify_completed": "OCI-applikation ändrad", "oci_modify_failed": "Ändring av OCI-applikation misslyckades", + "oci_watchdog_restarted": "OCI-applikation omstartad av watchdog", + "oci_watchdog_failed": "OCI-applikation stannar om och om igen", "app_update_available": "Appuppdatering tillgänglig", "lxc_update_applied": "LXC uppdatering tillämpad", "docker_stack_update_available": "Docker uppdateringar tillgängliga" @@ -6441,6 +6445,16 @@ "body": "Ändringen av {app_name} slutfördes inte.\nOrsak: {reason}\nContainrar: {containers}\nÖppna Hantera installerade OCI-applikationer för att kontrollera dess tillstånd.", "label": "Ändring av OCI-applikation misslyckades" }, + "oci_watchdog_restarted": { + "title": "{hostname}: {app_name} omstartad efter att ha stannat", + "body": "{app_name} stannade av sig själv och watchdog startade den igen.\nContainrar: {containers}", + "label": "OCI-applikation omstartad av watchdog" + }, + "oci_watchdog_failed": { + "title": "{hostname}: {app_name} stannar om och om igen", + "body": "{app_name} fortsätter att stanna av sig själv och watchdog startar inte längre om den.\nContainrar: {containers}\nStarta den manuellt när orsaken är löst; watchdog övervakar den sedan igen.", + "label": "OCI-applikation stannar om och om igen" + }, "app_update_available": { "title": "{hostname}: {app_name} uppdatering tillgänglig på CT {vmid}", "body": "{app_name} på CT {vmid} ({ct_name}) har en ny version:\n {installed} → {latest}", diff --git a/AppImage/scripts/flask_notification_routes.py b/AppImage/scripts/flask_notification_routes.py index ba605211..0e2c19cd 100644 --- a/AppImage/scripts/flask_notification_routes.py +++ b/AppImage/scripts/flask_notification_routes.py @@ -1631,12 +1631,13 @@ _OCI_EVENTS = { 'oci_update_completed': 'INFO', 'oci_update_failed': 'WARNING', 'oci_recreate_completed': 'INFO', 'oci_recreate_failed': 'WARNING', 'oci_modify_completed': 'INFO', 'oci_modify_failed': 'WARNING', + 'oci_watchdog_restarted': 'WARNING', 'oci_watchdog_failed': 'CRITICAL', } @notification_bp.route('/api/internal/oci-event', methods=['POST']) def internal_oci_event(): - """Called by the OCI engine when an update, a change or a recreation ends, with its + """Called by the OCI engine when an operation ends or the watchdog acts, with its result. Only accepts requests from this host.""" remote_addr = request.remote_addr or '' try: diff --git a/AppImage/scripts/flask_server.py b/AppImage/scripts/flask_server.py index 4e6e68f1..823a5e1a 100644 --- a/AppImage/scripts/flask_server.py +++ b/AppImage/scripts/flask_server.py @@ -15438,6 +15438,39 @@ def api_lxc_oci_instance(vmid): return jsonify({"ok": False, "error": str(e)}), 500 +@app.route('/api/lxc//oci-watchdog', methods=['POST']) +@require_admin_scope +def api_lxc_oci_watchdog(vmid): + """Turn the watchdog of an OCI application on or off: whether it is + started again when it stops on its own. The OCI engine keeps the choice + in the record of every container of the application. + + Body: {"enabled": true|false} + """ + data = request.get_json(silent=True) or {} + enabled = data.get('enabled') + if not isinstance(enabled, bool): + return jsonify({'error': 'enabled must be true or false'}), 400 + try: + import oci_instance_info + if not oci_instance_info.info(vmid).get('oci_instance'): + return jsonify({'error': 'not an OCI application'}), 404 + result = subprocess.run( + [sys.executable, '/usr/local/share/proxmenux/oci/engine/remote/oci_watchdog.py', + 'set', str(int(vmid)), 'on' if enabled else 'off'], + capture_output=True, text=True, timeout=60, + env={k: v for k, v in os.environ.items() if k not in ('PYTHONPATH', 'LD_LIBRARY_PATH')}) + if result.returncode == 3: + return jsonify({'error': 'busy', 'detail': 'Another OCI operation is in progress'}), 409 + if result.returncode != 0: + return jsonify({'error': (result.stderr or result.stdout).strip()[-300:] or 'failed'}), 500 + return jsonify({'ok': True, 'watchdog': enabled, **oci_instance_info.info(vmid)}) + except subprocess.TimeoutExpired: + return jsonify({'error': 'timeout'}), 504 + except Exception as e: + return jsonify({'ok': False, 'error': str(e)}), 500 + + @app.route('/api/vms//logs', methods=['GET']) @require_auth def api_vm_logs(vmid): diff --git a/AppImage/scripts/notification_templates.py b/AppImage/scripts/notification_templates.py index 6a12a0f5..bf9073fc 100644 --- a/AppImage/scripts/notification_templates.py +++ b/AppImage/scripts/notification_templates.py @@ -927,6 +927,24 @@ TEMPLATES = { 'group': 'vm_ct', 'default_enabled': True, }, + 'oci_watchdog_restarted': { + 'title': '{hostname}: {app_name} restarted after it stopped', + 'body': '{app_name} stopped on its own and the watchdog started it again.\nContainers: {containers}', + 'label': 'OCI application restarted by the watchdog', + 'group': 'vm_ct', + 'default_enabled': True, + }, + 'oci_watchdog_failed': { + 'title': '{hostname}: {app_name} keeps stopping', + 'body': ( + '{app_name} keeps stopping on its own and the watchdog no longer restarts it.\n' + 'Containers: {containers}\n' + 'Start it by hand once the cause is solved; the watchdog then watches it again.' + ), + 'label': 'OCI application keeps stopping', + 'group': 'vm_ct', + 'default_enabled': True, + }, 'app_update_available': { 'title': '{hostname}: {app_name} update available on CT {vmid}', 'body': ( @@ -2377,6 +2395,8 @@ EVENT_EMOJI = { 'oci_recreate_failed': '\u26A0\uFE0F', 'oci_modify_completed': '\u2705', 'oci_modify_failed': '\u26A0\uFE0F', + 'oci_watchdog_restarted': '\U0001F504', + 'oci_watchdog_failed': '\u26A0\uFE0F', 'app_update_available': '\U0001F195', # \ud83c\udd95 NEW \u2014 upstream app release 'docker_stack_update_available': '\U0001F433', 'vm_start': '\u25B6\uFE0F', # play button diff --git a/AppImage/scripts/oci_instance_info.py b/AppImage/scripts/oci_instance_info.py index f95e7f92..818eff91 100644 --- a/AppImage/scripts/oci_instance_info.py +++ b/AppImage/scripts/oci_instance_info.py @@ -3,8 +3,9 @@ Read-only view of the installation record OCI manager Apps keeps for every container it created: whether the container is one, whether it belongs to a multi-container application, whether it uses host directories (which its -backup does not revert), whether an operation is pending and whether it was -restored from a backup and is not registered on this host yet. Nothing here +backup does not revert), whether an operation is pending, whether it is +restarted when it stops on its own and whether it was restored from a backup +and is not registered on this host yet. Nothing here changes the record or runs inside the container. """ from __future__ import annotations @@ -70,6 +71,7 @@ def info(vmid: int) -> dict: "host_directories": False, "pending": False, "restored": False, + "watchdog": False, } record = _record(vmid) installation = _installation(vmid) @@ -91,5 +93,6 @@ def info(vmid: int) -> dict: members=members, host_directories=any(_host_dirs(r) for r in records), pending=bool(record.get("pending_transaction") or primary.get("pending_stack_transaction")), + watchdog=all((r.get("deployment") or {}).get("watchdog") is True for r in records), ) return result diff --git a/install_proxmenux.sh b/install_proxmenux.sh index d387c0e5..883b06ca 100755 --- a/install_proxmenux.sh +++ b/install_proxmenux.sh @@ -898,6 +898,7 @@ install_normal_version() { mkdir -p "$BASE_DIR/oci/engine" cp -r "./oci/"* "$BASE_DIR/oci/engine/" find "$BASE_DIR/oci/engine" -type f -name '*.sh' -exec chmod +x {} + + systemctl try-restart proxmenux-oci-watchdog.service >/dev/null 2>&1 || true fi chmod +x "$BASE_DIR/install_proxmenux.sh" msg_ok "Necessary files created." diff --git a/install_proxmenux_beta.sh b/install_proxmenux_beta.sh index f4a42059..3171de4e 100644 --- a/install_proxmenux_beta.sh +++ b/install_proxmenux_beta.sh @@ -788,6 +788,7 @@ install_beta() { mkdir -p "$BASE_DIR/oci/engine" cp -r "./oci/"* "$BASE_DIR/oci/engine/" find "$BASE_DIR/oci/engine" -type f -name '*.sh' -exec chmod +x {} + + systemctl try-restart proxmenux-oci-watchdog.service >/dev/null 2>&1 || true fi chmod +x "$INSTALL_DIR/$MENU_SCRIPT" [ -f "$BASE_DIR/install_proxmenux.sh" ] && chmod +x "$BASE_DIR/install_proxmenux.sh" diff --git a/lang/es.json b/lang/es.json index b3e9e1c3..cc1b4a18 100644 --- a/lang/es.json +++ b/lang/es.json @@ -4768,6 +4768,8 @@ "Purge the gasket-dkms package": "Purgar el paquete gasket-dkms", "Purging gasket-dkms package...": "Purgando el paquete gasket-dkms...", "Purging log2ram apt package...": "Purgando el paquete log2ram apt...", + "Put this application under watchdog? It is restarted automatically when it crashes.": "¿Poner esta aplicación bajo vigilancia? Se reinicia automáticamente cuando falla.", + "Put this application under watchdog? It is restarted automatically when it crashes. A stop or a shutdown you ask for is never undone.": "¿Poner esta aplicación bajo vigilancia? Se reinicia automáticamente cuando falla. Una parada o un apagado que pidas tú nunca se deshace.", "Pwndrop is a self-deployable file hosting service for sending out red teaming payloads or securely sharing your private files over HTTP and WebDAV.": "Pwndrop es un servicio de alojamiento de archivos que despliegas tú mismo, para distribuir payloads de red team o compartir de forma segura tus archivos privados por HTTP y WebDAV.", "PyCharm offers out-of-the-box support for Python, databases, Jupyter, Git, Conda, PyTorch, TensorFlow, Hugging Face, Django, Flask, FastAPI, and more.": "PyCharm ofrece soporte de serie para Python, bases de datos, Jupyter, Git, Conda, PyTorch, TensorFlow, Hugging Face, Django, Flask, FastAPI y más.", "Pydio Cells needs an external MySQL or MariaDB database. The setup wizard asks for its address, database name and user on the first start.": "Pydio Cells necesita una base de datos MySQL o MariaDB externa. El asistente de configuración pide su dirección, el nombre de la base de datos y el usuario en el primer arranque.", @@ -6517,6 +6519,7 @@ "The variable contains line breaks:": "La variable contiene roturas de línea:", "The vfio.conf entries have been removed and initramfs rebuilt.": "Las entradas de vfio.conf se eliminaron y se reconstruyó initramfs.", "The volume ID does not contain a disk name for this CT; automatic adoption is unsafe": "El ID del volumen no contiene un nombre de disco de este CT; la adopción automática no es segura", + "The watchdog could not be enabled. Turn it on from Manage installed OCI applications.": "No se pudo activar la vigilancia. Actívala desde Gestionar aplicaciones OCI instaladas.", "The web UI password must have at least 24 characters": "La contraseña de la interfaz web debe tener al menos 24 caracteres", "The web UI user contains characters that are not allowed": "El usuario de la interfaz web contiene caracteres no permitidos", "The web interface is served over plain HTTP on port 51821 (INSECURE=true). Keep it inside the local network or publish it through a reverse proxy with TLS.": "La interfaz web se sirve por HTTP sin cifrar en el puerto 51821 (INSECURE=true). Mantenla dentro de la red local o publícala a través de un proxy inverso con TLS.", @@ -6551,6 +6554,7 @@ "This application cannot be recovered yet:": "Esta aplicación todavía no se puede recuperar:", "This application cannot be recovered yet; nothing was changed:": "Esta aplicación todavía no se puede recuperar; no se ha cambiado nada:", "This application is not an Immich installed by ProxMenux": "Esta aplicación no es un Immich instalado por ProxMenux", + "This application is under watchdog: it is restarted automatically when it crashes. Turn the watchdog off?": "Esta aplicación está bajo vigilancia: se reinicia automáticamente cuando falla. ¿Desactivar la vigilancia?", "This archive does not contain a recognized backup layout.": "Este archivo no contiene un diseño de backup reconocido.", "This backup does not contain any restorable paths.": "Este backup no contiene rutas restaurables.", "This backup includes /etc/zfs/zpool.cache (host-specific ZFS state).": "Este backup incluye /etc/zfs/zpool.cache (estado ZFS específico del host).", @@ -7192,6 +7196,11 @@ "Warning: Limited PCI Reset Support": "Advertencia: soporte limitado para reinicio de PCI", "Warning: both VMs have autostart enabled (onboot=1).": "Advertencia: ambas máquinas virtuales tienen el inicio automático habilitado (onboot=1).", "Warnings": "Advertencias", + "Watchdog": "Vigilancia", + "Watchdog (off): restart the application when it crashes": "Vigilancia (desactivada): reiniciar la aplicación cuando falla", + "Watchdog (on): restart the application when it crashes": "Vigilancia (activada): reiniciar la aplicación cuando falla", + "Watchdog disabled.": "Vigilancia desactivada.", + "Watchdog enabled: the application is restarted when it crashes.": "Vigilancia activada: la aplicación se reinicia cuando falla.", "WeKnora is an LLM-powered framework designed for deep document understanding and semantic retrieval, especially for handling complex, heterogeneous documents.": "WeKnora es un framework basado en LLM diseñado para la comprensión profunda de documentos y la recuperación semántica, en especial con documentos complejos y heterogéneos.", "Web UI": "Web UI", "Web UI 1": "Web UI 1", diff --git a/oci/remote/oci_watchdog.py b/oci/remote/oci_watchdog.py new file mode 100644 index 00000000..a896d833 --- /dev/null +++ b/oci/remote/oci_watchdog.py @@ -0,0 +1,305 @@ +#!/usr/bin/env python3 +"""Start again an OCI application that stopped on its own. + +Proxmox has no restart policy: when the process of an application container +ends, the container stays stopped. This service watches the containers whose +record asks for it and starts again the one that stopped without anybody +asking: a stop, a shutdown, a backup, a migration or a ProxMenux operation is +never undone, and neither is a container that was already stopped when the +service first saw it. +""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path +import subprocess +import sys +import time + +import oci_instances as instances +import oci_operation_notice + +UNIT = 'proxmenux-oci-watchdog.service' +UNIT_FILE = Path('/etc/systemd/system') / UNIT +JOURNAL = Path('/usr/local/share/proxmenux/scripts/global/pmx_journal.sh') +CGROUPS = Path('/sys/fs/cgroup/lxc') +CONFIGS = Path('/etc/pve/lxc') +HA_RESOURCES = Path('/etc/pve/ha/resources.cfg') +TASKS = (Path('/var/log/pve/tasks/active'), Path('/var/log/pve/tasks/index')) +INTERVAL = 10 +# The clock of a task has one second of resolution. +TOLERANCE = 2 +# An application that ran this long before stopping is not in a crash loop: +# its restart, such as the one it asks for after saving its settings, is +# immediate however often it happens. +STABLE = 20 +# One notice of a restart per application in this time, however often it restarts. +NOTICE_EVERY = 3600 +# Wait before each new attempt; after the last one the application is left stopped. +DELAYS = (0, 30, 60, 120, 300) +# The marks of an operation are kept for the notices that arrive after it ends. +OPERATION_GRACE = 120 +STOP_TASKS = {'vzstop', 'vzshutdown', 'vzsuspend', 'vzreboot', 'vzdestroy', 'vzmigrate', 'vzrestore', 'vzdump'} +HOST_TASKS = {'stopall', 'migrateall'} + + +def new_state(): + return {'seen': None, 'since': None, 'attempts': 0, 'retry_at': 0.0, 'settled': False, 'noticed': None} + + +def notice_due(state, now): + """Whether to tell that the application was restarted: the first time, + and then at most once in a while.""" + if state['noticed'] is not None and now - state['noticed'] < NOTICE_EVERY: + return False + state['noticed'] = now + return True + + +def parse_tasks(text, active): + """The tasks of a Proxmox task list as (type, id, end); `end` is None + while the task runs. The list of active tasks carries one more column.""" + tasks = [] + for line in text.splitlines(): + fields = line.split() + if not fields or not fields[0].startswith('UPID:'): + continue + parts = fields[0].split(':') + if len(parts) < 8: + continue + stamp = fields[2:3] if active else fields[1:2] + try: + end = int(stamp[0], 16) if stamp else None + except ValueError: + continue + tasks.append((parts[5], parts[6], end)) + return tasks + + +def stop_requested(tasks, vmid, seen): + """Whether somebody asked for the container, or for every guest of the + host, to stop: the request is still running or ended after the container + was last seen running.""" + return any((kind in HOST_TASKS or (kind in STOP_TASKS and target == str(vmid))) + and (end is None or end >= seen - TOLERANCE) for kind, target, end in tasks) + + +def decide(state, running, now, busy, requested): + """One look at one container. Returns 'start' when it has to be started + again, 'give-up' when it keeps stopping, or None. `busy` and `requested` + are only called for a container that was running and no longer is.""" + if running: + if state['since'] is None: + state['since'] = now + if now - state['since'] >= STABLE: + state['attempts'] = 0 + state.update(seen=now, settled=False) + return None + state['since'] = None + if state['seen'] is None or state['settled']: + return None + if busy(): + return None + if requested(state['seen']): + state['settled'] = True + return None + if now < state['retry_at']: + return None + if state['attempts'] >= len(DELAYS): + state['settled'] = True + return 'give-up' + state['attempts'] += 1 + state['retry_at'] = now + (DELAYS[state['attempts']] if state['attempts'] < len(DELAYS) else DELAYS[-1]) + return 'start' + + +def watched(root): + """The installed applications that asked for the watchdog: {vmid: name}.""" + found = {} + for path in sorted(root.glob('*/oci-compose.json')): + try: + record = json.loads(path.read_text()) + vmid = int(record['vmid']) + except (OSError, ValueError, KeyError, TypeError): + continue + if record.get('status') != 'installed' or record.get('deployment', {}).get('watchdog') is not True: + continue + if record.get('pending_stack_transaction') or record.get('pending_transaction'): + continue + title = record.get('template', {}).get('catalog_ui', {}).get('title') + found[vmid] = (title.get('en_US') if isinstance(title, dict) else title) or f'CT {vmid}' + return found + + +def is_running(vmid): + return (CGROUPS / str(vmid)).is_dir() + + +def is_busy(vmid, now): + """Something is working on the container or on the host: look again later.""" + try: + config = (CONFIGS / f'{vmid}.conf').read_text(errors='replace') + except OSError: + return True + if any(line.startswith('lock:') for line in config.split('\n[', 1)[0].splitlines()): + return True + try: + mark = json.loads((oci_operation_notice.MARKERS / str(vmid)).read_text()) + if mark.get('ended') is None or now - float(mark['ended']) < OPERATION_GRACE: + return True + except (OSError, ValueError, TypeError, KeyError): + pass + try: + if any(line.split(':', 1)[0].strip() == 'ct' and line.split(':', 1)[1].strip() == str(vmid) + for line in HA_RESOURCES.read_text().splitlines() if ':' in line): + return True + except OSError: + pass + state = subprocess.run(['systemctl', 'is-system-running'], capture_output=True, text=True, check=False) + return state.stdout.strip() == 'stopping' + + +def read_tasks(): + tasks = [] + for path in TASKS: + try: + with path.open('rb') as handle: + handle.seek(0, 2) + handle.seek(max(0, handle.tell() - 65536)) + tasks += parse_tasks(handle.read().decode(errors='replace'), path.name == 'active') + except OSError: + continue + return tasks + + +def start(vmid): + # In a scope of its own: what Proxmox leaves running for the container, + # such as its DHCP client, must not end when this service is restarted. + result = subprocess.run(['systemd-run', '--scope', '--quiet', '--collect', 'pct', 'start', str(vmid)], + capture_output=True, text=True, check=False, timeout=300) + return result.returncode == 0 + + +def look(states, now): + apps = watched(instances.ROOT) + for vmid in set(states) - set(apps): + del states[vmid] + for vmid, name in apps.items(): + state = states.setdefault(vmid, new_state()) + action = decide(state, is_running(vmid), now, lambda: is_busy(vmid, now), + lambda seen: stop_requested(read_tasks(), vmid, seen)) + data = {'app_name': name, 'vmid': vmid, 'containers': f'CT {vmid}'} + if action == 'start': + print(f'CT {vmid} ({name}) stopped on its own; starting it again (attempt {state["attempts"]})', flush=True) + try: + started = start(vmid) + except (OSError, subprocess.SubprocessError): + started = False + if started and notice_due(state, now): + oci_operation_notice.notify('oci_watchdog_restarted', data) + elif action == 'give-up': + print(f'CT {vmid} ({name}) keeps stopping; it is left stopped', flush=True) + oci_operation_notice.notify('oci_watchdog_failed', data) + + +def run(): + states = {} + while True: + try: + look(states, time.time()) + except (OSError, ValueError) as error: + print(f'watchdog: {error}', file=sys.stderr, flush=True) + time.sleep(INTERVAL) + + +def _journaled(unit): + """Write and enable the unit through the change journal of ProxMenux, so + the Changes tab of the Monitor shows it. False when the journal is not + installed on this host.""" + if not JOURNAL.is_file(): + return False + script = ('source "$1" && pmx_journal_context "oci_watchdog" "1.0" "oci_watchdog.py" ' + '&& pmx_write_file "$2" && systemctl daemon-reload && pmx_enable_service "$3"') + result = subprocess.run(['bash', '-c', script, 'bash', str(JOURNAL), str(UNIT_FILE), UNIT], + input=unit, text=True, capture_output=True, check=False) + return result.returncode == 0 + + +def ensure_service(): + """Install the service and leave it running. A running one is only + restarted when its unit changed; the ProxMenux installer restarts it when + it replaces this program.""" + unit = ('[Unit]\n' + 'Description=ProxMenux OCI watchdog\n' + 'After=pve-guests.service\n\n' + '[Service]\n' + 'Type=simple\n' + f'ExecStart=/usr/bin/python3 {Path(__file__).resolve()} run\n' + 'Restart=on-failure\n' + 'RestartSec=30\n\n' + '[Install]\n' + 'WantedBy=multi-user.target\n') + changed = not UNIT_FILE.is_file() or UNIT_FILE.read_text() != unit + if not _journaled(unit): + if changed: + UNIT_FILE.write_text(unit) + subprocess.run(['systemctl', 'daemon-reload'], check=False) + subprocess.run(['systemctl', 'enable', UNIT], check=False, capture_output=True) + subprocess.run(['systemctl', 'restart' if changed else 'start', UNIT], check=False, capture_output=True) + + +def application(root, vmid): + """The containers of the application `vmid` belongs to: itself, or every + member of its stack.""" + record = instances.read(root, vmid) + primary_id = int((record.get('stack_member') or {}).get('primary_vmid') or vmid) + primary = record if primary_id == vmid else instances.read(root, primary_id) + members = [int(member['vmid']) for member in (primary.get('stack') or {}).get('members') or []] + return members or [vmid] + + +def set_watchdog(root, vmid, enabled): + """Turn the watchdog of an application on or off. The choice is kept in + the record of each of its containers, so updates and backups carry it.""" + import oci_carried_record + with instances.locked(root): + vmids = application(root, vmid) + for member in vmids: + record = instances.read(root, member) + record.setdefault('deployment', {})['watchdog'] = bool(enabled) + instances.write(instances.location(root, member), record) + oci_carried_record.carry(root, member, mount_stopped=False) + if enabled: + ensure_service() + return vmids + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + commands = parser.add_subparsers(dest='command', required=True) + commands.add_parser('run') + commands.add_parser('install') + change = commands.add_parser('set') + change.add_argument('vmid', type=int) + change.add_argument('state', choices=['on', 'off']) + args = parser.parse_args() + if args.command == 'install': + ensure_service() + elif args.command == 'set': + try: + vmids = set_watchdog(instances.ROOT, args.vmid, args.state == 'on') + except BlockingIOError: + print('Another OCI operation is using the instance registry.', file=sys.stderr) + return 3 + except (OSError, ValueError, KeyError) as error: + print(str(error) or type(error).__name__, file=sys.stderr) + return 1 + print(json.dumps({'vmids': vmids, 'watchdog': args.state == 'on'})) + else: + run() + return 0 + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/oci/src/proxmenux_oci/cli.py b/oci/src/proxmenux_oci/cli.py index ee39d158..5ce74def 100644 --- a/oci/src/proxmenux_oci/cli.py +++ b/oci/src/proxmenux_oci/cli.py @@ -346,6 +346,8 @@ def _deployment_summary_text(template: dict[str, Any], deployment: dict[str, Any lines.append("") row(translate("Start"), f"{translate('when finished')}: {_yes_no(plan.get('start_after_create'))} · " f"{translate('with Proxmox')}: {_yes_no(plan.get('onboot'))}") + if "watchdog" in plan: + row(translate("Watchdog"), _yes_no(plan["watchdog"])) return "\n".join(lines) @@ -383,6 +385,14 @@ def _print_installation_summary(result: dict[str, Any], images_removed: str | No # ---------------------------------------------------------------- menus +def watchdog_default(template: dict[str, Any]) -> bool: + """Whether the recipe of the application asks Docker to restart it.""" + policies = [template.get("container_contract", {}).get("restart")] + policies += [service.get("compose", {}).get("restart") + for service in template.get("compose_stack", {}).get("services", []) if isinstance(service, dict)] + return any(policy in ("always", "unless-stopped", "on-failure") for policy in policies) + + def _install(catalog: Catalog, ui, item: dict[str, Any], mode: str) -> None: install_template(ui, catalog.compose(item["id"]), item["id"], mode) @@ -397,6 +407,10 @@ def install_template(ui, template: dict[str, Any], identifier: str, mode: str) - candidate = copy.deepcopy(template) try: deployment = build_deployment(candidate, wizard, mode) + # Every installation decides it, the default one as well. + deployment["watchdog"] = wizard.confirm( + translate("Put this application under watchdog? It is restarted automatically when it crashes."), + watchdog_default(candidate)) approved = wizard.review(_deployment_summary_text(candidate, deployment), translate("Installation summary"), question=translate("Install with this configuration?")) @@ -413,6 +427,8 @@ def install_template(ui, template: dict[str, Any], identifier: str, mode: str) - if not approved: return None template = candidate + # The engine installs the application; the watchdog is turned on once it is there. + watchdog = bool(deployment.pop("watchdog", False)) console.show_logo() console.msg_title(f"{source_text(template['catalog_ui']['title']) or identifier} · {APP_TITLE}") try: @@ -427,6 +443,12 @@ def install_template(ui, template: dict[str, Any], identifier: str, mode: str) - vmids = {int(v) for v in [result.get("vmid"), *(result.get("stack_vmids") or {}).values()] if v} _, removed = images.offer_removal(ui, sorted(vmids)) _print_installation_summary(result, removed) + if watchdog: + from .management import set_watchdog + if set_watchdog(PROJECT_ROOT, sorted(vmids), True): + console.msg_ok(translate("Watchdog enabled: the application is restarted when it crashes.")) + else: + console.msg_warn(translate("The watchdog could not be enabled. Turn it on from Manage installed OCI applications.")) console.wait_for_enter(translate("Press Enter to return to the menu...")) return result diff --git a/oci/src/proxmenux_oci/management.py b/oci/src/proxmenux_oci/management.py index f0c27bf1..f2b90e4d 100644 --- a/oci/src/proxmenux_oci/management.py +++ b/oci/src/proxmenux_oci/management.py @@ -269,6 +269,39 @@ def _interactive_management(project, ui): manage_instance(project, ui, row) +def set_watchdog(project, vmids, enabled): + """Turn the watchdog of the applications these containers belong to on or + off. False when the registry is busy or a record cannot be read.""" + sys.path.insert(0, str(project / 'remote')) + import oci_instances as instances + import oci_watchdog + done = set() + for vmid in vmids: + if vmid in done: + continue + try: + done.update(oci_watchdog.set_watchdog(instances.ROOT, vmid, enabled)) + except (OSError, ValueError, KeyError): + return False + return True + + +def _toggle_watchdog(project, ui, vmid, enabled): + """Ask and change the watchdog of an application from its menu.""" + question = (translate('This application is under watchdog: it is restarted automatically when it crashes. Turn the watchdog off?') + if enabled else + translate('Put this application under watchdog? It is restarted automatically when it crashes. A stop or a shutdown you ask for is never undone.')) + if not ui.confirm(question, not enabled): + return False + if not set_watchdog(project, [vmid], not enabled): + ui.message(translate('Another OCI operation is using the instance registry. Wait for it to finish and open this menu again; no container is modified.'), + translate('OCI management')) + return False + ui.message(translate('Watchdog disabled.') if enabled else translate('Watchdog enabled: the application is restarted when it crashes.'), + translate('OCI management')) + return True + + def manage_instance(project, ui, row, action=None, lifecycle_args=()): """What the menu does with one instance once it is selected. `action` skips the choice of operation, as ProxMenux Monitor does; the extra @@ -292,15 +325,21 @@ def manage_instance(project, ui, row, action=None, lifecycle_args=()): if row['status'] != 'installed' or row['reason'] != 'matched': ui.message(translate('The instance identity or status must be reviewed before updating.'), translate('OCI management')) return False - if action is None: - action = ui.choose(translate('Manage OCI'), [('update', translate('Update the image with the saved configuration')), - ('modify', translate('Modify: edit resources, network, paths and GPU')), - ('remove', translate('Remove: delete the application and its containers'))], 'update') - if action is None: - return False sys.path.insert(0, str(project / 'remote')) import oci_instances as instances record = instances.read(instances.ROOT, row['vmid']) + watched = record.get('deployment', {}).get('watchdog') is True + watchdog_label = (translate('Watchdog (on): restart the application when it crashes') if watched + else translate('Watchdog (off): restart the application when it crashes')) + if action is None: + action = ui.choose(translate('Manage OCI'), [('update', translate('Update the image with the saved configuration')), + ('modify', translate('Modify: edit resources, network, paths and GPU')), + ('watchdog', watchdog_label), + ('remove', translate('Remove: delete the application and its containers'))], 'update') + if action is None: + return False + if action == 'watchdog': + return _toggle_watchdog(project, ui, row['vmid'], watched) if action == 'remove': return _remove(project, ui, row['vmid']) import oci_instance_reconcile as reconcile @@ -489,10 +528,15 @@ def _manage_stack(project, ui, row, action=None, lifecycle_args=()): options.append(('modify', translate('Modify extra paths and devices'))) if updatable: options.append(('recreate', translate('Recreate every container with its saved configuration'))) + watched = primary.get('deployment', {}).get('watchdog') is True + options.append(('watchdog', translate('Watchdog (on): restart the application when it crashes') if watched + else translate('Watchdog (off): restart the application when it crashes'))) options.append(('remove', translate('Remove: delete the application and its containers'))) action = ui.choose(translate('Manage OCI stack'), options, options[0][0]) if action is None: return False + if action == 'watchdog': + return _toggle_watchdog(project, ui, primary_id, primary.get('deployment', {}).get('watchdog') is True) if action == 'remove': return _remove(project, ui, primary_id) if action == 'modify': diff --git a/oci/tests/test_stack_recreate_v1.py b/oci/tests/test_stack_recreate_v1.py index 23aafda0..a104315f 100644 --- a/oci/tests/test_stack_recreate_v1.py +++ b/oci/tests/test_stack_recreate_v1.py @@ -53,6 +53,7 @@ class StackRecreateV1Tests(unittest.TestCase): "base_config_sha256": "saved-config"}} with patch.object(adapter, "validate"), \ patch.object(adapter, "state", return_value={"backups": {"138": {"archive": "/backup"}}}), \ + patch.object(native, "translate", side_effect=lambda text: text), \ patch.object(native.member_tx, "apply") as apply: adapter.replace(138, prepared, "transaction-id") diff --git a/oci/tests/test_watchdog.py b/oci/tests/test_watchdog.py new file mode 100644 index 00000000..92d0d4b7 --- /dev/null +++ b/oci/tests/test_watchdog.py @@ -0,0 +1,209 @@ +"""The watchdog starts again what stopped on its own, and nothing else.""" + +import json +from pathlib import Path +import sys +import tempfile +import unittest + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "remote")) + +import oci_watchdog as watchdog + +ACTIVE = """UPID:amd:001C34BC:01A809CF:6AC55103:vzshutdown:109:root@pam: 1 6AC55105 OK +UPID:amd:001C3370:01A80568:6AC550F8:vzstart:109:root@pam: 1 6AC550FA OK +UPID:amd:001C32AC:01A80211:6AC550EF:vzstop:108:root@pam: 1 6AC550F1 unable to stop: timeout +UPID:amd:001C3999:01A80999:6AC55200:vzdump:110:root@pam: 0 +""" +INDEX = """UPID:amd:001C32AC:01A80211:6AC550EF:vzstop:109:root@pam: 6AC550F1 OK +UPID:amd:001C3370:01A80568:6AC550F8:stopall::root@pam: 6AC55300 unable to stop every guest +""" + + +def never(): + raise AssertionError("not expected to be asked") + + +class Watch: + """One container seen through the decisions of the watchdog.""" + def __init__(self): + self.state = watchdog.new_state() + + def look(self, running, now, busy=False, requested=False): + return watchdog.decide(self.state, running, now, lambda: busy, lambda seen: requested) + + +class TaskList(unittest.TestCase): + def test_both_lists_are_read_with_their_own_columns(self): + self.assertEqual(watchdog.parse_tasks(ACTIVE, True), [ + ("vzshutdown", "109", 0x6AC55105), ("vzstart", "109", 0x6AC550FA), + ("vzstop", "108", 0x6AC550F1), ("vzdump", "110", None)]) + self.assertEqual(watchdog.parse_tasks(INDEX, False), [("vzstop", "109", 0x6AC550F1), ("stopall", "", 0x6AC55300)]) + + def test_a_request_counts_when_it_ended_after_the_container_was_last_seen_running(self): + tasks = watchdog.parse_tasks(ACTIVE, True) + self.assertTrue(watchdog.stop_requested(tasks, 109, 0x6AC55104)) + self.assertFalse(watchdog.stop_requested(tasks, 109, 0x6AC55110)) + self.assertFalse(watchdog.stop_requested(tasks, 111, 0x6AC55000)) + + def test_a_running_request_and_a_stop_of_every_guest_count(self): + self.assertTrue(watchdog.stop_requested(watchdog.parse_tasks(ACTIVE, True), 110, 0x6AC55900)) + self.assertTrue(watchdog.stop_requested(watchdog.parse_tasks(INDEX, False), 555, 0x6AC552F0)) + + def test_starting_a_container_is_not_a_request_to_stop_it(self): + self.assertFalse(watchdog.stop_requested([("vzstart", "109", None)], 109, 0)) + + +class Decisions(unittest.TestCase): + def test_a_container_never_seen_running_is_left_alone(self): + watch = Watch() + self.assertIsNone(watchdog.decide(watch.state, False, 100, never, never)) + self.assertIsNone(watchdog.decide(watch.state, False, 5000, never, never)) + + def test_a_container_that_stops_on_its_own_is_started_again(self): + watch = Watch() + watch.look(True, 100) + self.assertEqual(watch.look(False, 110), "start") + self.assertEqual(watch.state["attempts"], 1) + + def test_a_requested_stop_is_never_undone(self): + watch = Watch() + watch.look(True, 100) + self.assertIsNone(watch.look(False, 110, requested=True)) + self.assertIsNone(watchdog.decide(watch.state, False, 9000, never, never)) + watch.look(True, 9100) + self.assertEqual(watch.look(False, 9110), "start") + + def test_nothing_is_decided_while_something_works_on_the_container(self): + watch = Watch() + watch.look(True, 100) + self.assertIsNone(watchdog.decide(watch.state, False, 110, lambda: True, never)) + self.assertIsNone(watch.look(False, 120, requested=True)) + watch.look(True, 300) + self.assertIsNone(watchdog.decide(watch.state, False, 310, lambda: True, never)) + self.assertEqual(watch.look(False, 320), "start") + + def test_a_stop_asked_after_a_restart_is_respected(self): + watch = Watch() + watch.look(True, 100) + self.assertEqual(watch.look(False, 110), "start") + watch.look(True, 120) + self.assertIsNone(watch.look(False, 130, requested=True)) + self.assertIsNone(watchdog.decide(watch.state, False, 5000, never, never)) + + def test_an_application_that_keeps_stopping_is_tried_with_longer_waits_and_then_left(self): + watch, now, actions = Watch(), 100, [] + watch.look(True, now) + for _ in range(200): + now += 10 + action = watch.look(False, now) + if action: + actions.append((action, now)) + self.assertEqual([name for name, _ in actions], ["start"] * 5 + ["give-up"]) + waits = [later - earlier for (_, earlier), (_, later) in zip(actions, actions[1:])] + self.assertEqual(waits, [30, 60, 120, 300, 300]) + self.assertIsNone(watchdog.decide(watch.state, False, now + 9000, never, never)) + + def test_an_application_that_restarts_itself_now_and_then_is_always_started_at_once(self): + watch, now = Watch(), 100 + watch.look(True, now) + for _ in range(12): + now += 10 + self.assertEqual(watch.look(False, now), "start") + for _ in range(4): + now += 10 + watch.look(True, now) + self.assertEqual(watch.state["attempts"], 0) + + def test_a_restart_is_told_once_in_a_while(self): + state = watchdog.new_state() + self.assertTrue(watchdog.notice_due(state, 1000)) + self.assertFalse(watchdog.notice_due(state, 1000 + watchdog.NOTICE_EVERY - 1)) + self.assertTrue(watchdog.notice_due(state, 1000 + watchdog.NOTICE_EVERY)) + + def test_running_for_a_while_forgets_the_earlier_attempts(self): + watch = Watch() + watch.look(True, 100) + self.assertEqual(watch.look(False, 110), "start") + watch.look(True, 120) + watch.look(True, 120 + watchdog.STABLE) + self.assertEqual(watch.state["attempts"], 0) + self.assertEqual(watch.look(False, 400), "start") + self.assertEqual(watch.state["attempts"], 1) + + +class Registry(unittest.TestCase): + def test_only_installed_applications_that_asked_for_it_are_watched(self): + with tempfile.TemporaryDirectory() as folder: + root = Path(folder) + records = { + 101: {"status": "installed", "deployment": {"watchdog": True}, + "template": {"catalog_ui": {"title": {"en_US": "Jellyfin"}}}}, + 102: {"status": "installed", "deployment": {"watchdog": False}}, + 103: {"status": "installed", "deployment": {}}, + 104: {"status": "updating", "deployment": {"watchdog": True}}, + 105: {"status": "installed", "deployment": {"watchdog": True}, "pending_stack_transaction": "/journal"}, + 106: {"status": "installed", "deployment": {"watchdog": True}}, + } + for vmid, record in records.items(): + (root / str(vmid)).mkdir() + (root / str(vmid) / "oci-compose.json").write_text(json.dumps({"vmid": vmid, **record})) + (root / "107").mkdir() + (root / "107" / "oci-compose.json").write_text("{broken") + self.assertEqual(watchdog.watched(root), {101: "Jellyfin", 106: "CT 106"}) + + +class Start(unittest.TestCase): + def test_the_container_is_started_outside_the_service(self): + from unittest.mock import patch + with patch.object(watchdog.subprocess, "run") as run: + run.return_value.returncode = 0 + self.assertTrue(watchdog.start(109)) + run.return_value.returncode = 255 + self.assertFalse(watchdog.start(109)) + self.assertEqual(run.call_args.args[0], ["systemd-run", "--scope", "--quiet", "--collect", "pct", "start", "109"]) + + +class Switch(unittest.TestCase): + """Turning the watchdog on or off for a whole application.""" + def registry(self, folder): + root = Path(folder) + ids = {101: "11111111-1111-4111-8111-111111111111", 102: "22222222-2222-4222-8222-222222222222", + 103: "33333333-3333-4333-8333-333333333333"} + stack = {"members": [{"vmid": 101}, {"vmid": 102}]} + records = {101: {"stack": stack, "stack_member": {"primary_vmid": 101}}, + 102: {"stack_member": {"primary_vmid": 101}}, 103: {}} + for vmid, extra in records.items(): + (root / str(vmid)).mkdir() + (root / str(vmid) / "oci-compose.json").write_text(json.dumps({ + "schema_version": 1, "vmid": vmid, "installation_id": ids[vmid], "status": "installed", + "deployment": {"onboot": True}, **extra})) + return root + + def flags(self, root): + return {int(path.parent.name): json.loads(path.read_text())["deployment"].get("watchdog") + for path in root.glob("*/oci-compose.json")} + + def test_every_container_of_the_application_takes_the_choice_and_keeps_the_rest_of_its_record(self): + from unittest.mock import patch + import oci_carried_record + with tempfile.TemporaryDirectory() as folder: + root = self.registry(folder) + with patch.object(watchdog, "ensure_service") as service, \ + patch.object(oci_carried_record, "carry", return_value="carried") as carry: + self.assertEqual(watchdog.set_watchdog(root, 102, True), [101, 102]) + self.assertEqual(self.flags(root), {101: True, 102: True, 103: None}) + service.assert_called_once_with() + self.assertEqual([call.args[1] for call in carry.call_args_list], [101, 102]) + self.assertEqual(watchdog.set_watchdog(root, 103, True), [103]) + self.assertEqual(watchdog.set_watchdog(root, 101, False), [101, 102]) + self.assertEqual(self.flags(root), {101: False, 102: False, 103: True}) + self.assertEqual(service.call_count, 2) + record = json.loads((root / "101" / "oci-compose.json").read_text()) + self.assertEqual(record["deployment"]["onboot"], True) + self.assertEqual(record["stack"]["members"], [{"vmid": 101}, {"vmid": 102}]) + + +if __name__ == "__main__": + unittest.main() diff --git a/web/messages/en/docs/oci-manager/lifecycle.json b/web/messages/en/docs/oci-manager/lifecycle.json index cd119850..c328c0b9 100644 --- a/web/messages/en/docs/oci-manager/lifecycle.json +++ b/web/messages/en/docs/oci-manager/lifecycle.json @@ -49,6 +49,11 @@ "The editor opens with the current contract; the CT is rebuilt with the changes", "The data of container disks and host directories" ], + [ + "Watchdog: restart the application when it crashes", + "Whether the application is started again when it stops on its own. Nothing is rebuilt", + "The whole installation; only the choice is saved in the contract" + ], [ "Remove: delete the application and its containers", "The LXC, or every member of a stack, and their contracts are deleted", @@ -58,7 +63,7 @@ } }, { - "p": "For a multi-container application the menu offers Update every container of the application, Modify extra paths and devices, Recreate every container with its saved configuration and the removal. Modify adds or removes the extra paths and devices of the application container without rebuilding anything and, in Immich, changes what runs recognition. Recreate rebuilds every container from the image it was installed with, without looking for a newer one." + "p": "For a multi-container application the menu offers Update every container of the application, Modify extra paths and devices, Recreate every container with its saved configuration, the watchdog and the removal. Modify adds or removes the extra paths and devices of the application container without rebuilding anything and, in Immich, changes what runs recognition. Recreate rebuilds every container from the image it was installed with, without looking for a newer one." }, { "figure": { @@ -158,6 +163,52 @@ } ] }, + { + "id": "watchdog", + "title": "Watchdog: restarting an application that stops", + "intro": "Proxmox has no restart policy: when the process of an application container ends, the container stays stopped. With the watchdog on, an application that stops on its own is started again. It does for an OCI application what restart: unless-stopped does in a Compose file.", + "blocks": [ + { + "table": { + "headers": [ + "What happens", + "With the watchdog on" + ], + "rows": [ + [ + "The application crashes or its process ends", + "The container is started again within seconds" + ], + [ + "It is stopped or shut down from Proxmox, from the Monitor or with pct", + "It stays stopped until somebody starts it" + ], + [ + "A backup, a migration or a ProxMenux operation is working on it", + "Nothing is done until that ends" + ], + [ + "It was already stopped when the host started", + "It stays stopped; Start with Proxmox decides that" + ], + [ + "It keeps stopping right after starting", + "An application that stops again within seconds of starting is tried five times, waiting 30 seconds, 1, 2 and 5 minutes between them. Then it is left stopped and a notification is sent" + ] + ] + } + }, + { + "p": "The installer asks it in both installation modes and proposes what the recipe of the application declares. It is changed later from Manage installed OCI applications or with the Watchdog switch of the container in ProxMenux Monitor, and it applies to every container of a multi-container application. One service of the host, proxmenux-oci-watchdog.service, watches every application that asked for it; its installation is listed in the Changes tab of Audit & Report." + }, + { + "calloutInfo": { + "title": "What the host cannot tell", + "body": "The exit code of the application does not reach the host, so a clean exit and a crash look the same and both are restarted, as always and unless-stopped do in Docker. A stop asked outside Proxmox, with lxc-stop or systemctl, leaves no stop task and is taken as a stop of its own." + } + } + ] + }, { "id": "restore", "title": "Backup, restore and another host", diff --git a/web/messages/en/docs/oci-manager/monitor.json b/web/messages/en/docs/oci-manager/monitor.json index ba9e273e..99d3c5b5 100644 --- a/web/messages/en/docs/oci-manager/monitor.json +++ b/web/messages/en/docs/oci-manager/monitor.json @@ -25,6 +25,10 @@ "What changes for an OCI container" ], "rows": [ + [ + "Status", + "A Watchdog switch under Start at boot: the application is started again when it stops on its own" + ], [ "App", "The application is identified from the record, and updates are tracked by image" diff --git a/web/messages/es/docs/oci-manager/lifecycle.json b/web/messages/es/docs/oci-manager/lifecycle.json index 9fbca3c2..57ddbf9e 100644 --- a/web/messages/es/docs/oci-manager/lifecycle.json +++ b/web/messages/es/docs/oci-manager/lifecycle.json @@ -49,6 +49,11 @@ "El editor se abre con el contrato actual; el CT se reconstruye con los cambios", "Los datos de los discos del contenedor y de los directorios del host" ], + [ + "Vigilancia: reiniciar la aplicación cuando falla", + "Si la aplicación se inicia de nuevo cuando se detiene sola. No se reconstruye nada", + "Toda la instalación; solo se guarda la elección en el contrato" + ], [ "Eliminar: la aplicación y sus contenedores", "Se eliminan el LXC, o todos los miembros de una pila, y sus contratos", @@ -58,7 +63,7 @@ } }, { - "p": "Para una aplicación multicontenedor el menú ofrece Actualizar cada contenedor de la aplicación, Modificar rutas extra y dispositivos, Recrear todos los contenedores con su configuración guardada y la eliminación. Modificar añade o elimina las rutas y los dispositivos extra del contenedor de la aplicación sin reconstruir nada y, en Immich, cambia qué ejecuta el reconocimiento. Recrear reconstruye cada contenedor desde la imagen con la que se instaló, sin buscar una más nueva." + "p": "Para una aplicación multicontenedor el menú ofrece Actualizar cada contenedor de la aplicación, Modificar rutas extra y dispositivos, Recrear todos los contenedores con su configuración guardada, la vigilancia y la eliminación. Modificar añade o elimina las rutas y los dispositivos extra del contenedor de la aplicación sin reconstruir nada y, en Immich, cambia qué ejecuta el reconocimiento. Recrear reconstruye cada contenedor desde la imagen con la que se instaló, sin buscar una más nueva." }, { "figure": { @@ -158,6 +163,52 @@ } ] }, + { + "id": "watchdog", + "title": "Vigilancia: reiniciar una aplicación que se detiene", + "intro": "Proxmox no tiene política de reinicio: cuando termina el proceso de un contenedor de aplicación, el contenedor se queda parado. Con la vigilancia activada, una aplicación que se detiene sola se inicia de nuevo. Hace por una aplicación OCI lo que restart: unless-stopped hace en un fichero Compose.", + "blocks": [ + { + "table": { + "headers": [ + "Qué ocurre", + "Con la vigilancia activada" + ], + "rows": [ + [ + "La aplicación falla o su proceso termina", + "El contenedor se inicia de nuevo en segundos" + ], + [ + "Se detiene o se apaga desde Proxmox, desde el Monitor o con pct", + "Se queda parado hasta que alguien lo inicia" + ], + [ + "Un backup, una migración o una operación de ProxMenux está trabajando en él", + "No se hace nada hasta que termina" + ], + [ + "Ya estaba parado cuando arrancó el host", + "Se queda parado; eso lo decide Iniciar con Proxmox" + ], + [ + "No deja de detenerse nada más arrancar", + "Una aplicación que vuelve a detenerse a los pocos segundos de arrancar se intenta cinco veces, con esperas de 30 segundos, 1, 2 y 5 minutos entre ellas. Después se deja parada y se envía una notificación" + ] + ] + } + }, + { + "p": "El instalador lo pregunta en los dos modos de instalación y propone lo que declara la receta de la aplicación. Después se cambia desde Gestionar aplicaciones OCI instaladas o con el interruptor Vigilancia del contenedor en ProxMenux Monitor, y se aplica a todos los contenedores de una aplicación multicontenedor. Un único servicio del host, proxmenux-oci-watchdog.service, vigila todas las aplicaciones que lo pidieron; su instalación aparece en la pestaña Cambios de Audit & Report." + }, + { + "calloutInfo": { + "title": "Lo que el host no puede distinguir", + "body": "El código de salida de la aplicación no llega al host, así que una salida limpia y un fallo se ven igual y ambos se reinician, como hacen always y unless-stopped en Docker. Una parada pedida fuera de Proxmox, con lxc-stop o systemctl, no deja tarea de parada y se toma como una parada propia." + } + } + ] + }, { "id": "restore", "title": "Backup, restauración y otro host", diff --git a/web/messages/es/docs/oci-manager/monitor.json b/web/messages/es/docs/oci-manager/monitor.json index 6bcfe43f..80b1361b 100644 --- a/web/messages/es/docs/oci-manager/monitor.json +++ b/web/messages/es/docs/oci-manager/monitor.json @@ -25,6 +25,10 @@ "Qué cambia en un contenedor OCI" ], "rows": [ + [ + "Estado", + "Un interruptor Vigilancia bajo Iniciar al arrancar: la aplicación se inicia de nuevo cuando se detiene sola" + ], [ "App", "La aplicación se identifica desde el registro y sus actualizaciones se siguen por la imagen"