feat(oci): watchdog that restarts an application when it stops on its own

- New service that starts again an OCI application that stopped on its own, such as Frigate after Save & Restart. A stop or shutdown asked by the user, a backup, a migration and a ProxMenux operation are never undone.
- The installer asks it in both modes and proposes what the recipe declares; it is changed from the management menu or with the Watchdog switch of the container modal in the Monitor.
- Notifications when the watchdog restarts an application and when it keeps stopping.
- The service is recorded in the change journal, and the feature is documented.
This commit is contained in:
MacRimi
2026-10-06 23:18:26 +02:00
parent 6b70c3934e
commit 41ae752da1
28 changed files with 1006 additions and 19 deletions
@@ -7,6 +7,7 @@ from unittest import TestCase
ROOT = Path(__file__).resolve().parents[3]
LOCALES = ('en', 'es', 'de', 'fr', 'it', 'pt', 'sk', 'sv')
EVENTS = [f'oci_{kind}_{result}' for kind in ('update', 'modify', 'recreate') for result in ('completed', 'failed')]
EVENTS += ['oci_watchdog_restarted', 'oci_watchdog_failed']
class OciOperationEvents(TestCase):
@@ -17,7 +18,7 @@ class OciOperationEvents(TestCase):
for event in EVENTS:
self.assertIn(f"'{event}':", accepted, event)
self.assertEqual(len(re.findall(rf"^ '{event}': \{{$", templates, re.MULTILINE)), 1, event)
self.assertEqual(len(re.findall(rf"^ '{event}': +'\\u", templates, re.MULTILINE)), 1, event)
self.assertEqual(len(re.findall(rf"^ '{event}': +'\\[uU]", templates, re.MULTILINE)), 1, event)
def test_every_language_has_the_label_and_the_message_of_every_event(self):
fields = lambda text: sorted(re.findall(r'\{(\w+)\}', text))
@@ -69,7 +69,7 @@ class SelectionSetupWording(TestCase):
# It cannot be updated, but it can still be changed or removed.
ui = self._stack({'members': [{'native_stack_intent': {'adapt': True}}]}, REPLAY)
ui.choose.assert_called_once()
self.assertEqual([tag for tag, _ in ui.choose.call_args.args[1]], ['modify', 'remove'])
self.assertEqual([tag for tag, _ in ui.choose.call_args.args[1]], ['modify', 'watchdog', 'remove'])
def _stack(self, stack, expected):
record = {'stack': stack}
@@ -0,0 +1,80 @@
"""The watchdog of an OCI application is reachable from the Monitor and worded in every language."""
import ast
import json
from pathlib import Path
import re
from unittest import TestCase
ROOT = Path(__file__).resolve().parents[3]
LOCALES = ('en', 'es', 'de', 'fr', 'it', 'pt', 'sk', 'sv')
class OciWatchdogWiring(TestCase):
def test_changing_it_needs_an_administrator_and_goes_through_the_engine(self):
server = (ROOT / 'AppImage/scripts/flask_server.py').read_text()
route = re.search(r"@app\.route\('/api/lxc/<int:vmid>/oci-watchdog', methods=\['POST'\]\)\n@(\w+)\ndef (\w+)", server)
self.assertEqual(route.group(1), 'require_admin_scope')
body = server[route.end():server.index("@app.route('/api/vms/<int:vmid>/logs'")]
self.assertIn("if not isinstance(enabled, bool):", body)
self.assertIn("oci/engine/remote/oci_watchdog.py',", body)
self.assertNotIn('shell=True', body)
self.assertIn('"watchdog": False,', (ROOT / 'AppImage/scripts/oci_instance_info.py').read_text())
def test_the_toggle_is_only_offered_for_an_oci_application(self):
modal = (ROOT / 'AppImage/components/virtual-machines.tsx').read_text()
block = modal[modal.index('{/* Watchdog of an OCI application'):]
self.assertTrue(block.split('\n', 3)[2].strip().startswith('{ociInstance?.oci_instance && ('))
self.assertIn('t("vmLxc.details.watchdog")', block)
self.assertIn('`/api/lxc/${selectedVM.vmid}/oci-watchdog`', modal)
def test_every_language_names_and_explains_it(self):
for locale in LOCALES:
details = json.loads((ROOT / f'AppImage/messages/{locale}/common.json').read_text())['vmLxc']['details']
self.assertTrue(details['watchdog'].strip(), locale)
self.assertTrue(details['watchdogHelp'].strip(), locale)
def test_the_engine_restarts_its_service_when_proxmenux_replaces_it(self):
for installer in ('install_proxmenux.sh', 'install_proxmenux_beta.sh'):
self.assertIn('systemctl try-restart proxmenux-oci-watchdog.service', (ROOT / installer).read_text(), installer)
engine = (ROOT / 'oci/remote/oci_watchdog.py').read_text()
self.assertIn("UNIT = 'proxmenux-oci-watchdog.service'", engine)
class OciWatchdogInstaller(TestCase):
CLI = (ROOT / 'oci/src/proxmenux_oci/cli.py').read_text()
MENU = (ROOT / 'oci/src/proxmenux_oci/management.py').read_text()
def default(self, template):
node = next(item for item in ast.parse(self.CLI).body
if isinstance(item, ast.FunctionDef) and item.name == 'watchdog_default')
scope = {'Any': object}
exec(compile(ast.Module(body=[node], type_ignores=[]), 'cli.py', 'exec'), scope)
return scope['watchdog_default'](template)
def test_the_recipe_decides_what_is_proposed(self):
self.assertTrue(self.default({'container_contract': {'restart': 'unless-stopped'}}))
self.assertTrue(self.default({'container_contract': {'restart': 'always'}}))
self.assertTrue(self.default({'compose_stack': {'services': [{'compose': {}}, {'compose': {'restart': 'on-failure'}}]}}))
self.assertFalse(self.default({'container_contract': {'restart': None}}))
self.assertFalse(self.default({'container_contract': {'restart': 'no'}, 'compose_stack': {'services': [{'compose': {'restart': 'no'}}]}}))
self.assertFalse(self.default({}))
def test_every_installation_asks_it_and_the_engine_never_receives_the_answer(self):
install = self.CLI[self.CLI.index('def install_template('):self.CLI.index('def _rclone_mount(')]
asked = install.index('deployment["watchdog"] = wizard.confirm(')
self.assertLess(install.index('deployment = build_deployment(candidate, wizard, mode)'), asked)
self.assertLess(asked, install.index('approved = wizard.review('))
self.assertLess(install.index('watchdog = bool(deployment.pop("watchdog", False))'), install.index('run_remote_install('))
self.assertIn('set_watchdog(PROJECT_ROOT, sorted(vmids), True)', install)
self.assertIn('row(translate("Watchdog"), _yes_no(plan["watchdog"]))', self.CLI)
def test_both_management_menus_offer_it(self):
self.assertEqual(self.MENU.count("('watchdog', "), 2)
self.assertEqual(self.MENU.count("if action == 'watchdog':"), 2)
def test_the_service_is_written_through_the_change_journal(self):
engine = (ROOT / 'oci/remote/oci_watchdog.py').read_text()
self.assertIn('pmx_write_file "$2" && systemctl daemon-reload && pmx_enable_service "$3"', engine)
journal = (ROOT / 'scripts/global/pmx_journal.sh').read_text()
for helper in ('pmx_journal_context()', 'pmx_write_file()', 'pmx_enable_service()'):
self.assertIn(helper, journal)
+50 -5
View File
@@ -52,6 +52,7 @@ interface OciInstanceInfo {
host_directories: boolean
pending: boolean
restored?: boolean
watchdog?: boolean
}
interface LxcUpdateCheck {
@@ -887,6 +888,8 @@ export function VirtualMachines() {
// savedOnboot → 2 s ack pill after successful save.
const [resourcesEditMode, setResourcesEditMode] = useState(false)
const [pendingOnboot, setPendingOnboot] = useState<boolean | null>(null)
// Watchdog of an OCI application: same edit-gate as the toggle above.
const [pendingWatchdog, setPendingWatchdog] = useState<boolean | null>(null)
const [pendingTags, setPendingTags] = useState<string[] | null>(null)
const [newTagDraft, setNewTagDraft] = useState<string>("")
// When set (in edit mode), the pill at this index renders as an
@@ -1251,6 +1254,7 @@ export function VirtualMachines() {
// start in view mode with no pending change.
setResourcesEditMode(false)
setPendingOnboot(null)
setPendingWatchdog(null)
setPendingTags(null)
setNewTagDraft("")
setEditingTagIndex(null)
@@ -1682,6 +1686,7 @@ export function VirtualMachines() {
const handleCancelResourcesEdit = () => {
setPendingOnboot(null)
setPendingWatchdog(null)
setPendingTags(null)
setNewTagDraft("")
setEditingTagIndex(null)
@@ -1702,16 +1707,26 @@ export function VirtualMachines() {
if (pendingTags !== null && stringifyTags(pendingTags) !== stringifyTags(currentTags)) {
payload.tags = pendingTags
}
if (Object.keys(payload).length === 0) {
const watchdogChanged = !!ociInstance?.oci_instance && pendingWatchdog !== null && pendingWatchdog !== !!ociInstance.watchdog
if (Object.keys(payload).length === 0 && !watchdogChanged) {
handleCancelResourcesEdit()
return
}
setSavingOnboot(true)
try {
await fetchApi(`/api/vms/${selectedVM.vmid}/config`, {
method: "POST",
body: JSON.stringify(payload),
})
if (Object.keys(payload).length > 0) {
await fetchApi(`/api/vms/${selectedVM.vmid}/config`, {
method: "POST",
body: JSON.stringify(payload),
})
}
if (watchdogChanged) {
const updated = await fetchApi<OciInstanceInfo>(`/api/lxc/${selectedVM.vmid}/oci-watchdog`, {
method: "POST",
body: JSON.stringify({ enabled: pendingWatchdog }),
})
setOciInstance((prev) => (prev ? { ...prev, watchdog: !!updated?.watchdog } : prev))
}
// Optimistic local reflect so the UI doesn't wait a poll cycle.
if (payload.onboot !== undefined) {
setVMDetails((prev) => (prev ? { ...prev, config: { ...prev.config, onboot: payload.onboot as number } } : prev))
@@ -1725,6 +1740,7 @@ export function VirtualMachines() {
// list card too without waiting up to 2.5 s.
void mutate()
setPendingOnboot(null)
setPendingWatchdog(null)
setPendingTags(null)
setNewTagDraft("")
setEditingTagIndex(null)
@@ -4091,6 +4107,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
savingOnboot ||
(
(pendingOnboot === null || pendingOnboot === !!vmDetails.config.onboot) &&
(pendingWatchdog === null || !ociInstance?.oci_instance || pendingWatchdog === !!ociInstance.watchdog) &&
(pendingTags === null || stringifyTags(pendingTags) === stringifyTags(parseTags(selectedVM?.tags)))
)
}
@@ -4323,6 +4340,34 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
/>
</div>
{/* Watchdog of an OCI application: started
again when it stops on its own. */}
{ociInstance?.oci_instance && (
<div
className={`mt-1 rounded-md p-3 flex items-center justify-between gap-3 ${
resourcesEditMode ? "bg-accent" : ""
}`}
>
<div className="flex items-start gap-2 min-w-0">
<Eye className="h-4 w-4 text-blue-500 shrink-0 mt-0.5" />
<div className="min-w-0">
<div className="text-sm font-medium text-foreground">
{t("vmLxc.details.watchdog")}
</div>
<div className="text-xs text-muted-foreground leading-relaxed">
{t("vmLxc.details.watchdogHelp")}
</div>
</div>
</div>
<Switch
checked={pendingWatchdog ?? !!ociInstance.watchdog}
disabled={!resourcesEditMode || savingOnboot}
onCheckedChange={(v) => setPendingWatchdog(v)}
className={`data-[state=checked]:bg-blue-600 data-[state=unchecked]:bg-input border border-border ${!resourcesEditMode ? "opacity-60" : ""}`}
/>
</div>
)}
{/* IP Addresses with proper keys */}
{selectedVM?.type === "lxc" && vmDetails?.lxc_ip_info && (
<div className="mt-4 lg:mt-6">
+14
View File
@@ -1049,6 +1049,8 @@
"mounted": "montiert"
},
"startOnBoot": "Beim Booten beginnen",
"watchdog": "Watchdog",
"watchdogHelp": "Startet diese Anwendung automatisch neu, wenn sie abstürzt.",
"tags": "Tags",
"tagsPlaceholder": "Tag hinzufügen…",
"tagsNone": "Keine Tags",
@@ -2008,6 +2010,8 @@
"oci_recreate_failed": "Neuerstellung der OCI-Anwendung fehlgeschlagen",
"oci_modify_completed": "OCI-Anwendung geändert",
"oci_modify_failed": "Änderung der OCI-Anwendung fehlgeschlagen",
"oci_watchdog_restarted": "OCI-Anwendung vom Watchdog neu gestartet",
"oci_watchdog_failed": "OCI-Anwendung stoppt immer wieder",
"app_update_available": "App-Update verfügbar",
"lxc_update_applied": "LXC Update angewendet",
"docker_stack_update_available": "Docker Updates verfügbar"
@@ -6441,6 +6445,16 @@
"body": "Die Änderung von {app_name} wurde nicht abgeschlossen.\nGrund: {reason}\nContainer: {containers}\nÖffnen Sie „Installierte OCI-Anwendungen verwalten“, um den Zustand zu prüfen.",
"label": "Änderung der OCI-Anwendung fehlgeschlagen"
},
"oci_watchdog_restarted": {
"title": "{hostname}: {app_name} nach einem Stopp neu gestartet",
"body": "{app_name} wurde von selbst beendet und vom Watchdog neu gestartet.\nContainer: {containers}",
"label": "OCI-Anwendung vom Watchdog neu gestartet"
},
"oci_watchdog_failed": {
"title": "{hostname}: {app_name} stoppt immer wieder",
"body": "{app_name} stoppt immer wieder von selbst und wird vom Watchdog nicht mehr neu gestartet.\nContainer: {containers}\nStarten Sie die Anwendung von Hand, sobald die Ursache behoben ist; der Watchdog überwacht sie dann wieder.",
"label": "OCI-Anwendung stoppt immer wieder"
},
"app_update_available": {
"title": "{hostname}: {app_name} Update verfügbar auf CT {vmid}",
"body": "{app_name} auf CT {vmid} ({ct_name}) hat eine neue Version:\n {installed} → {latest}",
File diff suppressed because one or more lines are too long
+14
View File
@@ -1049,6 +1049,8 @@
"mounted": "montado"
},
"startOnBoot": "Iniciar al arrancar",
"watchdog": "Vigilancia",
"watchdogHelp": "Reinicia automáticamente esta aplicación cuando falla.",
"tags": "Etiquetas",
"tagsPlaceholder": "Añadir etiqueta…",
"tagsNone": "Sin etiquetas",
@@ -1995,6 +1997,8 @@
"oci_recreate_failed": "Recreación de aplicación OCI fallida",
"oci_modify_completed": "Aplicación OCI modificada",
"oci_modify_failed": "Modificación de aplicación OCI fallida",
"oci_watchdog_restarted": "Aplicación OCI reiniciada por la vigilancia",
"oci_watchdog_failed": "Aplicación OCI que se detiene una y otra vez",
"app_update_available": "Actualización de app disponible",
"ai_model_migrated": "Modelo de IA actualizado automáticamente",
"proxmenux_update": "Actualización de ProxMenux disponible",
@@ -6441,6 +6445,16 @@
"body": "La modificación de {app_name} no se completó.\nMotivo: {reason}\nContenedores: {containers}\nAbre Gestionar aplicaciones OCI instaladas para comprobar su estado.",
"label": "Modificación de aplicación OCI fallida"
},
"oci_watchdog_restarted": {
"title": "{hostname}: {app_name} reiniciado tras detenerse",
"body": "{app_name} se detuvo solo y la vigilancia lo ha iniciado de nuevo.\nContenedores: {containers}",
"label": "Aplicación OCI reiniciada por la vigilancia"
},
"oci_watchdog_failed": {
"title": "{hostname}: {app_name} se detiene una y otra vez",
"body": "{app_name} sigue deteniéndose solo y la vigilancia ya no lo reinicia.\nContenedores: {containers}\nInícialo a mano cuando la causa esté resuelta; la vigilancia volverá a vigilarlo.",
"label": "Aplicación OCI que se detiene una y otra vez"
},
"app_update_available": {
"title": "{hostname}: actualización de {app_name} disponible en CT {vmid}",
"body": "{app_name} en CT {vmid} ({ct_name}) tiene una nueva versión:\n {installed} → {latest}",
+14
View File
@@ -1049,6 +1049,8 @@
"mounted": "monté"
},
"startOnBoot": "Démarrer au démarrage",
"watchdog": "Surveillance",
"watchdogHelp": "Redémarre automatiquement cette application lorsqu'elle plante.",
"tags": "balises",
"tagsPlaceholder": "Ajouter une balise…",
"tagsNone": "Aucune balise",
@@ -2008,6 +2010,8 @@
"oci_recreate_failed": "Échec de la recréation de l'application OCI",
"oci_modify_completed": "Application OCI modifiée",
"oci_modify_failed": "Échec de la modification de l'application OCI",
"oci_watchdog_restarted": "Application OCI redémarrée par la surveillance",
"oci_watchdog_failed": "Application OCI qui s'arrête sans cesse",
"app_update_available": "mise à jour de l'application disponible",
"lxc_update_applied": "Mise à jour LXC appliquée",
"docker_stack_update_available": "Docker mises à jour disponibles"
@@ -6441,6 +6445,16 @@
"body": "La modification de {app_name} n'a pas abouti.\nMotif : {reason}\nConteneurs : {containers}\nOuvrez Gérer les applications OCI installées pour vérifier son état.",
"label": "Échec de la modification de l'application OCI"
},
"oci_watchdog_restarted": {
"title": "{hostname} : {app_name} redémarré après un arrêt",
"body": "{app_name} s'est arrêté tout seul et la surveillance l'a redémarré.\nConteneurs : {containers}",
"label": "Application OCI redémarrée par la surveillance"
},
"oci_watchdog_failed": {
"title": "{hostname} : {app_name} s'arrête sans cesse",
"body": "{app_name} continue de s'arrêter tout seul et la surveillance ne le redémarre plus.\nConteneurs : {containers}\nDémarrez-le à la main une fois la cause résolue ; la surveillance le reprend alors.",
"label": "Application OCI qui s'arrête sans cesse"
},
"app_update_available": {
"title": "{hostname} : mise à jour {app_name} disponible sur CT {vmid}",
"body": "{app_name} sur CT {vmid} ({ct_name}) a une nouvelle version :\n {installed} → {latest}",
+14
View File
@@ -1049,6 +1049,8 @@
"mounted": "montato"
},
"startOnBoot": "Avvio al boot",
"watchdog": "Watchdog",
"watchdogHelp": "Riavvia automaticamente questa applicazione quando si arresta in modo anomalo.",
"tags": "Tag",
"tagsPlaceholder": "Aggiungi tag…",
"tagsNone": "Nessun tag",
@@ -2008,6 +2010,8 @@
"oci_recreate_failed": "Ricreazione app OCI non riuscita",
"oci_modify_completed": "App OCI modificata",
"oci_modify_failed": "Modifica app OCI non riuscita",
"oci_watchdog_restarted": "App OCI riavviata dal watchdog",
"oci_watchdog_failed": "App OCI che continua ad arrestarsi",
"app_update_available": "Aggiornamento app disponibile",
"lxc_update_applied": "Aggiornamento LXC applicato",
"docker_stack_update_available": "Aggiornamenti Docker disponibili"
@@ -6441,6 +6445,16 @@
"body": "La modifica di {app_name} non è stata completata.\nMotivo: {reason}\nContainer: {containers}\nApri «Gestisci le applicazioni OCI installate» per verificarne lo stato.",
"label": "Modifica app OCI non riuscita"
},
"oci_watchdog_restarted": {
"title": "{hostname}: {app_name} riavviato dopo un arresto",
"body": "{app_name} si è arrestato da solo e il watchdog lo ha riavviato.\nContainer: {containers}",
"label": "App OCI riavviata dal watchdog"
},
"oci_watchdog_failed": {
"title": "{hostname}: {app_name} continua ad arrestarsi",
"body": "{app_name} continua ad arrestarsi da solo e il watchdog non lo riavvia più.\nContainer: {containers}\nAvvialo a mano una volta risolta la causa; il watchdog tornerà a sorvegliarlo.",
"label": "App OCI che continua ad arrestarsi"
},
"app_update_available": {
"title": "{hostname}: aggiornamento {app_name} disponibile su CT {vmid}",
"body": "{app_name} su CT {vmid} ({ct_name}) ha una nuova versione:\n {installed} → {latest}",
+14
View File
@@ -1049,6 +1049,8 @@
"mounted": "montado"
},
"startOnBoot": "Iniciar na inicialização",
"watchdog": "Vigilância",
"watchdogHelp": "Reinicia automaticamente esta aplicação quando falha.",
"tags": "Etiquetas",
"tagsPlaceholder": "Adicionar tag…",
"tagsNone": "sem tags",
@@ -2008,6 +2010,8 @@
"oci_recreate_failed": "Falha na recriação da aplicação OCI",
"oci_modify_completed": "Aplicação OCI modificada",
"oci_modify_failed": "Falha na modificação da aplicação OCI",
"oci_watchdog_restarted": "Aplicação OCI reiniciada pela vigilância",
"oci_watchdog_failed": "Aplicação OCI que continua a parar",
"app_update_available": "atualização de aplicativo disponível",
"lxc_update_applied": "Atualização LXC aplicada",
"docker_stack_update_available": "Docker atualizações disponíveis"
@@ -6441,6 +6445,16 @@
"body": "A modificação de {app_name} não foi concluída.\nMotivo: {reason}\nContêineres: {containers}\nAbra Gerir aplicações OCI instaladas para verificar o estado.",
"label": "Falha na modificação da aplicação OCI"
},
"oci_watchdog_restarted": {
"title": "{hostname}: {app_name} reiniciado depois de parar",
"body": "{app_name} parou sozinho e a vigilância voltou a iniciá-lo.\nContêineres: {containers}",
"label": "Aplicação OCI reiniciada pela vigilância"
},
"oci_watchdog_failed": {
"title": "{hostname}: {app_name} continua a parar",
"body": "{app_name} continua a parar sozinho e a vigilância já não o reinicia.\nContêineres: {containers}\nInicie-o à mão quando a causa estiver resolvida; a vigilância volta então a vigiá-lo.",
"label": "Aplicação OCI que continua a parar"
},
"app_update_available": {
"title": "{hostname}: atualização {app_name} disponível no CT {vmid}",
"body": "{app_name} no CT {vmid} ({ct_name}) tem uma nova versão:\n {installed} → {latest}",
+14
View File
@@ -1073,6 +1073,8 @@
"configuredButNotMounted": "Nastavené, ale nepripojené"
},
"startOnBoot": "Spúšťať pri štarte",
"watchdog": "Watchdog",
"watchdogHelp": "Automaticky reštartuje túto aplikáciu, keď zlyhá.",
"tags": "Značky",
"tagsPlaceholder": "Pridať značku…",
"tagsNone": "Žiadne značky"
@@ -2009,6 +2011,8 @@
"oci_recreate_failed": "Opätovné vytvorenie OCI aplikácie zlyhalo",
"oci_modify_completed": "OCI aplikácia upravená",
"oci_modify_failed": "Úprava OCI aplikácie zlyhala",
"oci_watchdog_restarted": "OCI aplikácia reštartovaná watchdogom",
"oci_watchdog_failed": "OCI aplikácia sa opakovane zastavuje",
"app_update_available": "K dispozícii je aktualizácia aplikácie"
},
"ui": {
@@ -6440,6 +6444,16 @@
"body": "Úprava aplikácie {app_name} sa nedokončila.\nDôvod: {reason}\nKontajnery: {containers}\nOtvorte Spravovať nainštalované OCI aplikácie a skontrolujte jej stav.",
"label": "Úprava OCI aplikácie zlyhala"
},
"oci_watchdog_restarted": {
"title": "{hostname}: {app_name} reštartovaná po zastavení",
"body": "Aplikácia {app_name} sa sama zastavila a watchdog ju znova spustil.\nKontajnery: {containers}",
"label": "OCI aplikácia reštartovaná watchdogom"
},
"oci_watchdog_failed": {
"title": "{hostname}: {app_name} sa opakovane zastavuje",
"body": "Aplikácia {app_name} sa stále sama zastavuje a watchdog ju už nereštartuje.\nKontajnery: {containers}\nPo odstránení príčiny ju spustite ručne; watchdog ju potom bude znova sledovať.",
"label": "OCI aplikácia sa opakovane zastavuje"
},
"app_update_available": {
"title": "{hostname}: Pre {app_name} je na CT {vmid} dostupná aktualizácia",
"body": "Aplikácia {app_name} na CT {vmid} ({ct_name}) má novú verziu:\n {installed} → {latest}",
+14
View File
@@ -1049,6 +1049,8 @@
"mounted": "monterad"
},
"startOnBoot": "Börja vid start",
"watchdog": "Watchdog",
"watchdogHelp": "Startar om den här applikationen automatiskt när den kraschar.",
"tags": "Taggar",
"tagsPlaceholder": "Lägg till tagg...",
"tagsNone": "Inga taggar",
@@ -2008,6 +2010,8 @@
"oci_recreate_failed": "Återskapande av OCI-applikation misslyckades",
"oci_modify_completed": "OCI-applikation ändrad",
"oci_modify_failed": "Ändring av OCI-applikation misslyckades",
"oci_watchdog_restarted": "OCI-applikation omstartad av watchdog",
"oci_watchdog_failed": "OCI-applikation stannar om och om igen",
"app_update_available": "Appuppdatering tillgänglig",
"lxc_update_applied": "LXC uppdatering tillämpad",
"docker_stack_update_available": "Docker uppdateringar tillgängliga"
@@ -6441,6 +6445,16 @@
"body": "Ändringen av {app_name} slutfördes inte.\nOrsak: {reason}\nContainrar: {containers}\nÖppna Hantera installerade OCI-applikationer för att kontrollera dess tillstånd.",
"label": "Ändring av OCI-applikation misslyckades"
},
"oci_watchdog_restarted": {
"title": "{hostname}: {app_name} omstartad efter att ha stannat",
"body": "{app_name} stannade av sig själv och watchdog startade den igen.\nContainrar: {containers}",
"label": "OCI-applikation omstartad av watchdog"
},
"oci_watchdog_failed": {
"title": "{hostname}: {app_name} stannar om och om igen",
"body": "{app_name} fortsätter att stanna av sig själv och watchdog startar inte längre om den.\nContainrar: {containers}\nStarta den manuellt när orsaken är löst; watchdog övervakar den sedan igen.",
"label": "OCI-applikation stannar om och om igen"
},
"app_update_available": {
"title": "{hostname}: {app_name} uppdatering tillgänglig på CT {vmid}",
"body": "{app_name} på CT {vmid} ({ct_name}) har en ny version:\n {installed} → {latest}",
@@ -1631,12 +1631,13 @@ _OCI_EVENTS = {
'oci_update_completed': 'INFO', 'oci_update_failed': 'WARNING',
'oci_recreate_completed': 'INFO', 'oci_recreate_failed': 'WARNING',
'oci_modify_completed': 'INFO', 'oci_modify_failed': 'WARNING',
'oci_watchdog_restarted': 'WARNING', 'oci_watchdog_failed': 'CRITICAL',
}
@notification_bp.route('/api/internal/oci-event', methods=['POST'])
def internal_oci_event():
"""Called by the OCI engine when an update, a change or a recreation ends, with its
"""Called by the OCI engine when an operation ends or the watchdog acts, with its
result. Only accepts requests from this host."""
remote_addr = request.remote_addr or ''
try:
+33
View File
@@ -15438,6 +15438,39 @@ def api_lxc_oci_instance(vmid):
return jsonify({"ok": False, "error": str(e)}), 500
@app.route('/api/lxc/<int:vmid>/oci-watchdog', methods=['POST'])
@require_admin_scope
def api_lxc_oci_watchdog(vmid):
"""Turn the watchdog of an OCI application on or off: whether it is
started again when it stops on its own. The OCI engine keeps the choice
in the record of every container of the application.
Body: {"enabled": true|false}
"""
data = request.get_json(silent=True) or {}
enabled = data.get('enabled')
if not isinstance(enabled, bool):
return jsonify({'error': 'enabled must be true or false'}), 400
try:
import oci_instance_info
if not oci_instance_info.info(vmid).get('oci_instance'):
return jsonify({'error': 'not an OCI application'}), 404
result = subprocess.run(
[sys.executable, '/usr/local/share/proxmenux/oci/engine/remote/oci_watchdog.py',
'set', str(int(vmid)), 'on' if enabled else 'off'],
capture_output=True, text=True, timeout=60,
env={k: v for k, v in os.environ.items() if k not in ('PYTHONPATH', 'LD_LIBRARY_PATH')})
if result.returncode == 3:
return jsonify({'error': 'busy', 'detail': 'Another OCI operation is in progress'}), 409
if result.returncode != 0:
return jsonify({'error': (result.stderr or result.stdout).strip()[-300:] or 'failed'}), 500
return jsonify({'ok': True, 'watchdog': enabled, **oci_instance_info.info(vmid)})
except subprocess.TimeoutExpired:
return jsonify({'error': 'timeout'}), 504
except Exception as e:
return jsonify({'ok': False, 'error': str(e)}), 500
@app.route('/api/vms/<int:vmid>/logs', methods=['GET'])
@require_auth
def api_vm_logs(vmid):
@@ -927,6 +927,24 @@ TEMPLATES = {
'group': 'vm_ct',
'default_enabled': True,
},
'oci_watchdog_restarted': {
'title': '{hostname}: {app_name} restarted after it stopped',
'body': '{app_name} stopped on its own and the watchdog started it again.\nContainers: {containers}',
'label': 'OCI application restarted by the watchdog',
'group': 'vm_ct',
'default_enabled': True,
},
'oci_watchdog_failed': {
'title': '{hostname}: {app_name} keeps stopping',
'body': (
'{app_name} keeps stopping on its own and the watchdog no longer restarts it.\n'
'Containers: {containers}\n'
'Start it by hand once the cause is solved; the watchdog then watches it again.'
),
'label': 'OCI application keeps stopping',
'group': 'vm_ct',
'default_enabled': True,
},
'app_update_available': {
'title': '{hostname}: {app_name} update available on CT {vmid}',
'body': (
@@ -2377,6 +2395,8 @@ EVENT_EMOJI = {
'oci_recreate_failed': '\u26A0\uFE0F',
'oci_modify_completed': '\u2705',
'oci_modify_failed': '\u26A0\uFE0F',
'oci_watchdog_restarted': '\U0001F504',
'oci_watchdog_failed': '\u26A0\uFE0F',
'app_update_available': '\U0001F195', # \ud83c\udd95 NEW \u2014 upstream app release
'docker_stack_update_available': '\U0001F433',
'vm_start': '\u25B6\uFE0F', # play button
+5 -2
View File
@@ -3,8 +3,9 @@
Read-only view of the installation record OCI manager Apps keeps for every
container it created: whether the container is one, whether it belongs to a
multi-container application, whether it uses host directories (which its
backup does not revert), whether an operation is pending and whether it was
restored from a backup and is not registered on this host yet. Nothing here
backup does not revert), whether an operation is pending, whether it is
restarted when it stops on its own and whether it was restored from a backup
and is not registered on this host yet. Nothing here
changes the record or runs inside the container.
"""
from __future__ import annotations
@@ -70,6 +71,7 @@ def info(vmid: int) -> dict:
"host_directories": False,
"pending": False,
"restored": False,
"watchdog": False,
}
record = _record(vmid)
installation = _installation(vmid)
@@ -91,5 +93,6 @@ def info(vmid: int) -> dict:
members=members,
host_directories=any(_host_dirs(r) for r in records),
pending=bool(record.get("pending_transaction") or primary.get("pending_stack_transaction")),
watchdog=all((r.get("deployment") or {}).get("watchdog") is True for r in records),
)
return result
+1
View File
@@ -898,6 +898,7 @@ install_normal_version() {
mkdir -p "$BASE_DIR/oci/engine"
cp -r "./oci/"* "$BASE_DIR/oci/engine/"
find "$BASE_DIR/oci/engine" -type f -name '*.sh' -exec chmod +x {} +
systemctl try-restart proxmenux-oci-watchdog.service >/dev/null 2>&1 || true
fi
chmod +x "$BASE_DIR/install_proxmenux.sh"
msg_ok "Necessary files created."
+1
View File
@@ -788,6 +788,7 @@ install_beta() {
mkdir -p "$BASE_DIR/oci/engine"
cp -r "./oci/"* "$BASE_DIR/oci/engine/"
find "$BASE_DIR/oci/engine" -type f -name '*.sh' -exec chmod +x {} +
systemctl try-restart proxmenux-oci-watchdog.service >/dev/null 2>&1 || true
fi
chmod +x "$INSTALL_DIR/$MENU_SCRIPT"
[ -f "$BASE_DIR/install_proxmenux.sh" ] && chmod +x "$BASE_DIR/install_proxmenux.sh"
+9
View File
@@ -4768,6 +4768,8 @@
"Purge the gasket-dkms package": "Purgar el paquete gasket-dkms",
"Purging gasket-dkms package...": "Purgando el paquete gasket-dkms...",
"Purging log2ram apt package...": "Purgando el paquete log2ram apt...",
"Put this application under watchdog? It is restarted automatically when it crashes.": "¿Poner esta aplicación bajo vigilancia? Se reinicia automáticamente cuando falla.",
"Put this application under watchdog? It is restarted automatically when it crashes. A stop or a shutdown you ask for is never undone.": "¿Poner esta aplicación bajo vigilancia? Se reinicia automáticamente cuando falla. Una parada o un apagado que pidas tú nunca se deshace.",
"Pwndrop is a self-deployable file hosting service for sending out red teaming payloads or securely sharing your private files over HTTP and WebDAV.": "Pwndrop es un servicio de alojamiento de archivos que despliegas tú mismo, para distribuir payloads de red team o compartir de forma segura tus archivos privados por HTTP y WebDAV.",
"PyCharm offers out-of-the-box support for Python, databases, Jupyter, Git, Conda, PyTorch, TensorFlow, Hugging Face, Django, Flask, FastAPI, and more.": "PyCharm ofrece soporte de serie para Python, bases de datos, Jupyter, Git, Conda, PyTorch, TensorFlow, Hugging Face, Django, Flask, FastAPI y más.",
"Pydio Cells needs an external MySQL or MariaDB database. The setup wizard asks for its address, database name and user on the first start.": "Pydio Cells necesita una base de datos MySQL o MariaDB externa. El asistente de configuración pide su dirección, el nombre de la base de datos y el usuario en el primer arranque.",
@@ -6517,6 +6519,7 @@
"The variable contains line breaks:": "La variable contiene roturas de línea:",
"The vfio.conf entries have been removed and initramfs rebuilt.": "Las entradas de vfio.conf se eliminaron y se reconstruyó initramfs.",
"The volume ID does not contain a disk name for this CT; automatic adoption is unsafe": "El ID del volumen no contiene un nombre de disco de este CT; la adopción automática no es segura",
"The watchdog could not be enabled. Turn it on from Manage installed OCI applications.": "No se pudo activar la vigilancia. Actívala desde Gestionar aplicaciones OCI instaladas.",
"The web UI password must have at least 24 characters": "La contraseña de la interfaz web debe tener al menos 24 caracteres",
"The web UI user contains characters that are not allowed": "El usuario de la interfaz web contiene caracteres no permitidos",
"The web interface is served over plain HTTP on port 51821 (INSECURE=true). Keep it inside the local network or publish it through a reverse proxy with TLS.": "La interfaz web se sirve por HTTP sin cifrar en el puerto 51821 (INSECURE=true). Mantenla dentro de la red local o publícala a través de un proxy inverso con TLS.",
@@ -6551,6 +6554,7 @@
"This application cannot be recovered yet:": "Esta aplicación todavía no se puede recuperar:",
"This application cannot be recovered yet; nothing was changed:": "Esta aplicación todavía no se puede recuperar; no se ha cambiado nada:",
"This application is not an Immich installed by ProxMenux": "Esta aplicación no es un Immich instalado por ProxMenux",
"This application is under watchdog: it is restarted automatically when it crashes. Turn the watchdog off?": "Esta aplicación está bajo vigilancia: se reinicia automáticamente cuando falla. ¿Desactivar la vigilancia?",
"This archive does not contain a recognized backup layout.": "Este archivo no contiene un diseño de backup reconocido.",
"This backup does not contain any restorable paths.": "Este backup no contiene rutas restaurables.",
"This backup includes /etc/zfs/zpool.cache (host-specific ZFS state).": "Este backup incluye /etc/zfs/zpool.cache (estado ZFS específico del host).",
@@ -7192,6 +7196,11 @@
"Warning: Limited PCI Reset Support": "Advertencia: soporte limitado para reinicio de PCI",
"Warning: both VMs have autostart enabled (onboot=1).": "Advertencia: ambas máquinas virtuales tienen el inicio automático habilitado (onboot=1).",
"Warnings": "Advertencias",
"Watchdog": "Vigilancia",
"Watchdog (off): restart the application when it crashes": "Vigilancia (desactivada): reiniciar la aplicación cuando falla",
"Watchdog (on): restart the application when it crashes": "Vigilancia (activada): reiniciar la aplicación cuando falla",
"Watchdog disabled.": "Vigilancia desactivada.",
"Watchdog enabled: the application is restarted when it crashes.": "Vigilancia activada: la aplicación se reinicia cuando falla.",
"WeKnora is an LLM-powered framework designed for deep document understanding and semantic retrieval, especially for handling complex, heterogeneous documents.": "WeKnora es un framework basado en LLM diseñado para la comprensión profunda de documentos y la recuperación semántica, en especial con documentos complejos y heterogéneos.",
"Web UI": "Web UI",
"Web UI 1": "Web UI 1",
+305
View File
@@ -0,0 +1,305 @@
#!/usr/bin/env python3
"""Start again an OCI application that stopped on its own.
Proxmox has no restart policy: when the process of an application container
ends, the container stays stopped. This service watches the containers whose
record asks for it and starts again the one that stopped without anybody
asking: a stop, a shutdown, a backup, a migration or a ProxMenux operation is
never undone, and neither is a container that was already stopped when the
service first saw it.
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
import subprocess
import sys
import time
import oci_instances as instances
import oci_operation_notice
UNIT = 'proxmenux-oci-watchdog.service'
UNIT_FILE = Path('/etc/systemd/system') / UNIT
JOURNAL = Path('/usr/local/share/proxmenux/scripts/global/pmx_journal.sh')
CGROUPS = Path('/sys/fs/cgroup/lxc')
CONFIGS = Path('/etc/pve/lxc')
HA_RESOURCES = Path('/etc/pve/ha/resources.cfg')
TASKS = (Path('/var/log/pve/tasks/active'), Path('/var/log/pve/tasks/index'))
INTERVAL = 10
# The clock of a task has one second of resolution.
TOLERANCE = 2
# An application that ran this long before stopping is not in a crash loop:
# its restart, such as the one it asks for after saving its settings, is
# immediate however often it happens.
STABLE = 20
# One notice of a restart per application in this time, however often it restarts.
NOTICE_EVERY = 3600
# Wait before each new attempt; after the last one the application is left stopped.
DELAYS = (0, 30, 60, 120, 300)
# The marks of an operation are kept for the notices that arrive after it ends.
OPERATION_GRACE = 120
STOP_TASKS = {'vzstop', 'vzshutdown', 'vzsuspend', 'vzreboot', 'vzdestroy', 'vzmigrate', 'vzrestore', 'vzdump'}
HOST_TASKS = {'stopall', 'migrateall'}
def new_state():
return {'seen': None, 'since': None, 'attempts': 0, 'retry_at': 0.0, 'settled': False, 'noticed': None}
def notice_due(state, now):
"""Whether to tell that the application was restarted: the first time,
and then at most once in a while."""
if state['noticed'] is not None and now - state['noticed'] < NOTICE_EVERY:
return False
state['noticed'] = now
return True
def parse_tasks(text, active):
"""The tasks of a Proxmox task list as (type, id, end); `end` is None
while the task runs. The list of active tasks carries one more column."""
tasks = []
for line in text.splitlines():
fields = line.split()
if not fields or not fields[0].startswith('UPID:'):
continue
parts = fields[0].split(':')
if len(parts) < 8:
continue
stamp = fields[2:3] if active else fields[1:2]
try:
end = int(stamp[0], 16) if stamp else None
except ValueError:
continue
tasks.append((parts[5], parts[6], end))
return tasks
def stop_requested(tasks, vmid, seen):
"""Whether somebody asked for the container, or for every guest of the
host, to stop: the request is still running or ended after the container
was last seen running."""
return any((kind in HOST_TASKS or (kind in STOP_TASKS and target == str(vmid)))
and (end is None or end >= seen - TOLERANCE) for kind, target, end in tasks)
def decide(state, running, now, busy, requested):
"""One look at one container. Returns 'start' when it has to be started
again, 'give-up' when it keeps stopping, or None. `busy` and `requested`
are only called for a container that was running and no longer is."""
if running:
if state['since'] is None:
state['since'] = now
if now - state['since'] >= STABLE:
state['attempts'] = 0
state.update(seen=now, settled=False)
return None
state['since'] = None
if state['seen'] is None or state['settled']:
return None
if busy():
return None
if requested(state['seen']):
state['settled'] = True
return None
if now < state['retry_at']:
return None
if state['attempts'] >= len(DELAYS):
state['settled'] = True
return 'give-up'
state['attempts'] += 1
state['retry_at'] = now + (DELAYS[state['attempts']] if state['attempts'] < len(DELAYS) else DELAYS[-1])
return 'start'
def watched(root):
"""The installed applications that asked for the watchdog: {vmid: name}."""
found = {}
for path in sorted(root.glob('*/oci-compose.json')):
try:
record = json.loads(path.read_text())
vmid = int(record['vmid'])
except (OSError, ValueError, KeyError, TypeError):
continue
if record.get('status') != 'installed' or record.get('deployment', {}).get('watchdog') is not True:
continue
if record.get('pending_stack_transaction') or record.get('pending_transaction'):
continue
title = record.get('template', {}).get('catalog_ui', {}).get('title')
found[vmid] = (title.get('en_US') if isinstance(title, dict) else title) or f'CT {vmid}'
return found
def is_running(vmid):
return (CGROUPS / str(vmid)).is_dir()
def is_busy(vmid, now):
"""Something is working on the container or on the host: look again later."""
try:
config = (CONFIGS / f'{vmid}.conf').read_text(errors='replace')
except OSError:
return True
if any(line.startswith('lock:') for line in config.split('\n[', 1)[0].splitlines()):
return True
try:
mark = json.loads((oci_operation_notice.MARKERS / str(vmid)).read_text())
if mark.get('ended') is None or now - float(mark['ended']) < OPERATION_GRACE:
return True
except (OSError, ValueError, TypeError, KeyError):
pass
try:
if any(line.split(':', 1)[0].strip() == 'ct' and line.split(':', 1)[1].strip() == str(vmid)
for line in HA_RESOURCES.read_text().splitlines() if ':' in line):
return True
except OSError:
pass
state = subprocess.run(['systemctl', 'is-system-running'], capture_output=True, text=True, check=False)
return state.stdout.strip() == 'stopping'
def read_tasks():
tasks = []
for path in TASKS:
try:
with path.open('rb') as handle:
handle.seek(0, 2)
handle.seek(max(0, handle.tell() - 65536))
tasks += parse_tasks(handle.read().decode(errors='replace'), path.name == 'active')
except OSError:
continue
return tasks
def start(vmid):
# In a scope of its own: what Proxmox leaves running for the container,
# such as its DHCP client, must not end when this service is restarted.
result = subprocess.run(['systemd-run', '--scope', '--quiet', '--collect', 'pct', 'start', str(vmid)],
capture_output=True, text=True, check=False, timeout=300)
return result.returncode == 0
def look(states, now):
apps = watched(instances.ROOT)
for vmid in set(states) - set(apps):
del states[vmid]
for vmid, name in apps.items():
state = states.setdefault(vmid, new_state())
action = decide(state, is_running(vmid), now, lambda: is_busy(vmid, now),
lambda seen: stop_requested(read_tasks(), vmid, seen))
data = {'app_name': name, 'vmid': vmid, 'containers': f'CT {vmid}'}
if action == 'start':
print(f'CT {vmid} ({name}) stopped on its own; starting it again (attempt {state["attempts"]})', flush=True)
try:
started = start(vmid)
except (OSError, subprocess.SubprocessError):
started = False
if started and notice_due(state, now):
oci_operation_notice.notify('oci_watchdog_restarted', data)
elif action == 'give-up':
print(f'CT {vmid} ({name}) keeps stopping; it is left stopped', flush=True)
oci_operation_notice.notify('oci_watchdog_failed', data)
def run():
states = {}
while True:
try:
look(states, time.time())
except (OSError, ValueError) as error:
print(f'watchdog: {error}', file=sys.stderr, flush=True)
time.sleep(INTERVAL)
def _journaled(unit):
"""Write and enable the unit through the change journal of ProxMenux, so
the Changes tab of the Monitor shows it. False when the journal is not
installed on this host."""
if not JOURNAL.is_file():
return False
script = ('source "$1" && pmx_journal_context "oci_watchdog" "1.0" "oci_watchdog.py" '
'&& pmx_write_file "$2" && systemctl daemon-reload && pmx_enable_service "$3"')
result = subprocess.run(['bash', '-c', script, 'bash', str(JOURNAL), str(UNIT_FILE), UNIT],
input=unit, text=True, capture_output=True, check=False)
return result.returncode == 0
def ensure_service():
"""Install the service and leave it running. A running one is only
restarted when its unit changed; the ProxMenux installer restarts it when
it replaces this program."""
unit = ('[Unit]\n'
'Description=ProxMenux OCI watchdog\n'
'After=pve-guests.service\n\n'
'[Service]\n'
'Type=simple\n'
f'ExecStart=/usr/bin/python3 {Path(__file__).resolve()} run\n'
'Restart=on-failure\n'
'RestartSec=30\n\n'
'[Install]\n'
'WantedBy=multi-user.target\n')
changed = not UNIT_FILE.is_file() or UNIT_FILE.read_text() != unit
if not _journaled(unit):
if changed:
UNIT_FILE.write_text(unit)
subprocess.run(['systemctl', 'daemon-reload'], check=False)
subprocess.run(['systemctl', 'enable', UNIT], check=False, capture_output=True)
subprocess.run(['systemctl', 'restart' if changed else 'start', UNIT], check=False, capture_output=True)
def application(root, vmid):
"""The containers of the application `vmid` belongs to: itself, or every
member of its stack."""
record = instances.read(root, vmid)
primary_id = int((record.get('stack_member') or {}).get('primary_vmid') or vmid)
primary = record if primary_id == vmid else instances.read(root, primary_id)
members = [int(member['vmid']) for member in (primary.get('stack') or {}).get('members') or []]
return members or [vmid]
def set_watchdog(root, vmid, enabled):
"""Turn the watchdog of an application on or off. The choice is kept in
the record of each of its containers, so updates and backups carry it."""
import oci_carried_record
with instances.locked(root):
vmids = application(root, vmid)
for member in vmids:
record = instances.read(root, member)
record.setdefault('deployment', {})['watchdog'] = bool(enabled)
instances.write(instances.location(root, member), record)
oci_carried_record.carry(root, member, mount_stopped=False)
if enabled:
ensure_service()
return vmids
def main():
parser = argparse.ArgumentParser(description=__doc__)
commands = parser.add_subparsers(dest='command', required=True)
commands.add_parser('run')
commands.add_parser('install')
change = commands.add_parser('set')
change.add_argument('vmid', type=int)
change.add_argument('state', choices=['on', 'off'])
args = parser.parse_args()
if args.command == 'install':
ensure_service()
elif args.command == 'set':
try:
vmids = set_watchdog(instances.ROOT, args.vmid, args.state == 'on')
except BlockingIOError:
print('Another OCI operation is using the instance registry.', file=sys.stderr)
return 3
except (OSError, ValueError, KeyError) as error:
print(str(error) or type(error).__name__, file=sys.stderr)
return 1
print(json.dumps({'vmids': vmids, 'watchdog': args.state == 'on'}))
else:
run()
return 0
if __name__ == '__main__':
raise SystemExit(main())
+22
View File
@@ -346,6 +346,8 @@ def _deployment_summary_text(template: dict[str, Any], deployment: dict[str, Any
lines.append("")
row(translate("Start"), f"{translate('when finished')}: {_yes_no(plan.get('start_after_create'))} · "
f"{translate('with Proxmox')}: {_yes_no(plan.get('onboot'))}")
if "watchdog" in plan:
row(translate("Watchdog"), _yes_no(plan["watchdog"]))
return "\n".join(lines)
@@ -383,6 +385,14 @@ def _print_installation_summary(result: dict[str, Any], images_removed: str | No
# ---------------------------------------------------------------- menus
def watchdog_default(template: dict[str, Any]) -> bool:
"""Whether the recipe of the application asks Docker to restart it."""
policies = [template.get("container_contract", {}).get("restart")]
policies += [service.get("compose", {}).get("restart")
for service in template.get("compose_stack", {}).get("services", []) if isinstance(service, dict)]
return any(policy in ("always", "unless-stopped", "on-failure") for policy in policies)
def _install(catalog: Catalog, ui, item: dict[str, Any], mode: str) -> None:
install_template(ui, catalog.compose(item["id"]), item["id"], mode)
@@ -397,6 +407,10 @@ def install_template(ui, template: dict[str, Any], identifier: str, mode: str) -
candidate = copy.deepcopy(template)
try:
deployment = build_deployment(candidate, wizard, mode)
# Every installation decides it, the default one as well.
deployment["watchdog"] = wizard.confirm(
translate("Put this application under watchdog? It is restarted automatically when it crashes."),
watchdog_default(candidate))
approved = wizard.review(_deployment_summary_text(candidate, deployment),
translate("Installation summary"),
question=translate("Install with this configuration?"))
@@ -413,6 +427,8 @@ def install_template(ui, template: dict[str, Any], identifier: str, mode: str) -
if not approved:
return None
template = candidate
# The engine installs the application; the watchdog is turned on once it is there.
watchdog = bool(deployment.pop("watchdog", False))
console.show_logo()
console.msg_title(f"{source_text(template['catalog_ui']['title']) or identifier} · {APP_TITLE}")
try:
@@ -427,6 +443,12 @@ def install_template(ui, template: dict[str, Any], identifier: str, mode: str) -
vmids = {int(v) for v in [result.get("vmid"), *(result.get("stack_vmids") or {}).values()] if v}
_, removed = images.offer_removal(ui, sorted(vmids))
_print_installation_summary(result, removed)
if watchdog:
from .management import set_watchdog
if set_watchdog(PROJECT_ROOT, sorted(vmids), True):
console.msg_ok(translate("Watchdog enabled: the application is restarted when it crashes."))
else:
console.msg_warn(translate("The watchdog could not be enabled. Turn it on from Manage installed OCI applications."))
console.wait_for_enter(translate("Press Enter to return to the menu..."))
return result
+50 -6
View File
@@ -269,6 +269,39 @@ def _interactive_management(project, ui):
manage_instance(project, ui, row)
def set_watchdog(project, vmids, enabled):
"""Turn the watchdog of the applications these containers belong to on or
off. False when the registry is busy or a record cannot be read."""
sys.path.insert(0, str(project / 'remote'))
import oci_instances as instances
import oci_watchdog
done = set()
for vmid in vmids:
if vmid in done:
continue
try:
done.update(oci_watchdog.set_watchdog(instances.ROOT, vmid, enabled))
except (OSError, ValueError, KeyError):
return False
return True
def _toggle_watchdog(project, ui, vmid, enabled):
"""Ask and change the watchdog of an application from its menu."""
question = (translate('This application is under watchdog: it is restarted automatically when it crashes. Turn the watchdog off?')
if enabled else
translate('Put this application under watchdog? It is restarted automatically when it crashes. A stop or a shutdown you ask for is never undone.'))
if not ui.confirm(question, not enabled):
return False
if not set_watchdog(project, [vmid], not enabled):
ui.message(translate('Another OCI operation is using the instance registry. Wait for it to finish and open this menu again; no container is modified.'),
translate('OCI management'))
return False
ui.message(translate('Watchdog disabled.') if enabled else translate('Watchdog enabled: the application is restarted when it crashes.'),
translate('OCI management'))
return True
def manage_instance(project, ui, row, action=None, lifecycle_args=()):
"""What the menu does with one instance once it is selected. `action`
skips the choice of operation, as ProxMenux Monitor does; the extra
@@ -292,15 +325,21 @@ def manage_instance(project, ui, row, action=None, lifecycle_args=()):
if row['status'] != 'installed' or row['reason'] != 'matched':
ui.message(translate('The instance identity or status must be reviewed before updating.'), translate('OCI management'))
return False
if action is None:
action = ui.choose(translate('Manage OCI'), [('update', translate('Update the image with the saved configuration')),
('modify', translate('Modify: edit resources, network, paths and GPU')),
('remove', translate('Remove: delete the application and its containers'))], 'update')
if action is None:
return False
sys.path.insert(0, str(project / 'remote'))
import oci_instances as instances
record = instances.read(instances.ROOT, row['vmid'])
watched = record.get('deployment', {}).get('watchdog') is True
watchdog_label = (translate('Watchdog (on): restart the application when it crashes') if watched
else translate('Watchdog (off): restart the application when it crashes'))
if action is None:
action = ui.choose(translate('Manage OCI'), [('update', translate('Update the image with the saved configuration')),
('modify', translate('Modify: edit resources, network, paths and GPU')),
('watchdog', watchdog_label),
('remove', translate('Remove: delete the application and its containers'))], 'update')
if action is None:
return False
if action == 'watchdog':
return _toggle_watchdog(project, ui, row['vmid'], watched)
if action == 'remove':
return _remove(project, ui, row['vmid'])
import oci_instance_reconcile as reconcile
@@ -489,10 +528,15 @@ def _manage_stack(project, ui, row, action=None, lifecycle_args=()):
options.append(('modify', translate('Modify extra paths and devices')))
if updatable:
options.append(('recreate', translate('Recreate every container with its saved configuration')))
watched = primary.get('deployment', {}).get('watchdog') is True
options.append(('watchdog', translate('Watchdog (on): restart the application when it crashes') if watched
else translate('Watchdog (off): restart the application when it crashes')))
options.append(('remove', translate('Remove: delete the application and its containers')))
action = ui.choose(translate('Manage OCI stack'), options, options[0][0])
if action is None:
return False
if action == 'watchdog':
return _toggle_watchdog(project, ui, primary_id, primary.get('deployment', {}).get('watchdog') is True)
if action == 'remove':
return _remove(project, ui, primary_id)
if action == 'modify':
+1
View File
@@ -53,6 +53,7 @@ class StackRecreateV1Tests(unittest.TestCase):
"base_config_sha256": "saved-config"}}
with patch.object(adapter, "validate"), \
patch.object(adapter, "state", return_value={"backups": {"138": {"archive": "/backup"}}}), \
patch.object(native, "translate", side_effect=lambda text: text), \
patch.object(native.member_tx, "apply") as apply:
adapter.replace(138, prepared, "transaction-id")
+209
View File
@@ -0,0 +1,209 @@
"""The watchdog starts again what stopped on its own, and nothing else."""
import json
from pathlib import Path
import sys
import tempfile
import unittest
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
import oci_watchdog as watchdog
ACTIVE = """UPID:amd:001C34BC:01A809CF:6AC55103:vzshutdown:109:root@pam: 1 6AC55105 OK
UPID:amd:001C3370:01A80568:6AC550F8:vzstart:109:root@pam: 1 6AC550FA OK
UPID:amd:001C32AC:01A80211:6AC550EF:vzstop:108:root@pam: 1 6AC550F1 unable to stop: timeout
UPID:amd:001C3999:01A80999:6AC55200:vzdump:110:root@pam: 0
"""
INDEX = """UPID:amd:001C32AC:01A80211:6AC550EF:vzstop:109:root@pam: 6AC550F1 OK
UPID:amd:001C3370:01A80568:6AC550F8:stopall::root@pam: 6AC55300 unable to stop every guest
"""
def never():
raise AssertionError("not expected to be asked")
class Watch:
"""One container seen through the decisions of the watchdog."""
def __init__(self):
self.state = watchdog.new_state()
def look(self, running, now, busy=False, requested=False):
return watchdog.decide(self.state, running, now, lambda: busy, lambda seen: requested)
class TaskList(unittest.TestCase):
def test_both_lists_are_read_with_their_own_columns(self):
self.assertEqual(watchdog.parse_tasks(ACTIVE, True), [
("vzshutdown", "109", 0x6AC55105), ("vzstart", "109", 0x6AC550FA),
("vzstop", "108", 0x6AC550F1), ("vzdump", "110", None)])
self.assertEqual(watchdog.parse_tasks(INDEX, False), [("vzstop", "109", 0x6AC550F1), ("stopall", "", 0x6AC55300)])
def test_a_request_counts_when_it_ended_after_the_container_was_last_seen_running(self):
tasks = watchdog.parse_tasks(ACTIVE, True)
self.assertTrue(watchdog.stop_requested(tasks, 109, 0x6AC55104))
self.assertFalse(watchdog.stop_requested(tasks, 109, 0x6AC55110))
self.assertFalse(watchdog.stop_requested(tasks, 111, 0x6AC55000))
def test_a_running_request_and_a_stop_of_every_guest_count(self):
self.assertTrue(watchdog.stop_requested(watchdog.parse_tasks(ACTIVE, True), 110, 0x6AC55900))
self.assertTrue(watchdog.stop_requested(watchdog.parse_tasks(INDEX, False), 555, 0x6AC552F0))
def test_starting_a_container_is_not_a_request_to_stop_it(self):
self.assertFalse(watchdog.stop_requested([("vzstart", "109", None)], 109, 0))
class Decisions(unittest.TestCase):
def test_a_container_never_seen_running_is_left_alone(self):
watch = Watch()
self.assertIsNone(watchdog.decide(watch.state, False, 100, never, never))
self.assertIsNone(watchdog.decide(watch.state, False, 5000, never, never))
def test_a_container_that_stops_on_its_own_is_started_again(self):
watch = Watch()
watch.look(True, 100)
self.assertEqual(watch.look(False, 110), "start")
self.assertEqual(watch.state["attempts"], 1)
def test_a_requested_stop_is_never_undone(self):
watch = Watch()
watch.look(True, 100)
self.assertIsNone(watch.look(False, 110, requested=True))
self.assertIsNone(watchdog.decide(watch.state, False, 9000, never, never))
watch.look(True, 9100)
self.assertEqual(watch.look(False, 9110), "start")
def test_nothing_is_decided_while_something_works_on_the_container(self):
watch = Watch()
watch.look(True, 100)
self.assertIsNone(watchdog.decide(watch.state, False, 110, lambda: True, never))
self.assertIsNone(watch.look(False, 120, requested=True))
watch.look(True, 300)
self.assertIsNone(watchdog.decide(watch.state, False, 310, lambda: True, never))
self.assertEqual(watch.look(False, 320), "start")
def test_a_stop_asked_after_a_restart_is_respected(self):
watch = Watch()
watch.look(True, 100)
self.assertEqual(watch.look(False, 110), "start")
watch.look(True, 120)
self.assertIsNone(watch.look(False, 130, requested=True))
self.assertIsNone(watchdog.decide(watch.state, False, 5000, never, never))
def test_an_application_that_keeps_stopping_is_tried_with_longer_waits_and_then_left(self):
watch, now, actions = Watch(), 100, []
watch.look(True, now)
for _ in range(200):
now += 10
action = watch.look(False, now)
if action:
actions.append((action, now))
self.assertEqual([name for name, _ in actions], ["start"] * 5 + ["give-up"])
waits = [later - earlier for (_, earlier), (_, later) in zip(actions, actions[1:])]
self.assertEqual(waits, [30, 60, 120, 300, 300])
self.assertIsNone(watchdog.decide(watch.state, False, now + 9000, never, never))
def test_an_application_that_restarts_itself_now_and_then_is_always_started_at_once(self):
watch, now = Watch(), 100
watch.look(True, now)
for _ in range(12):
now += 10
self.assertEqual(watch.look(False, now), "start")
for _ in range(4):
now += 10
watch.look(True, now)
self.assertEqual(watch.state["attempts"], 0)
def test_a_restart_is_told_once_in_a_while(self):
state = watchdog.new_state()
self.assertTrue(watchdog.notice_due(state, 1000))
self.assertFalse(watchdog.notice_due(state, 1000 + watchdog.NOTICE_EVERY - 1))
self.assertTrue(watchdog.notice_due(state, 1000 + watchdog.NOTICE_EVERY))
def test_running_for_a_while_forgets_the_earlier_attempts(self):
watch = Watch()
watch.look(True, 100)
self.assertEqual(watch.look(False, 110), "start")
watch.look(True, 120)
watch.look(True, 120 + watchdog.STABLE)
self.assertEqual(watch.state["attempts"], 0)
self.assertEqual(watch.look(False, 400), "start")
self.assertEqual(watch.state["attempts"], 1)
class Registry(unittest.TestCase):
def test_only_installed_applications_that_asked_for_it_are_watched(self):
with tempfile.TemporaryDirectory() as folder:
root = Path(folder)
records = {
101: {"status": "installed", "deployment": {"watchdog": True},
"template": {"catalog_ui": {"title": {"en_US": "Jellyfin"}}}},
102: {"status": "installed", "deployment": {"watchdog": False}},
103: {"status": "installed", "deployment": {}},
104: {"status": "updating", "deployment": {"watchdog": True}},
105: {"status": "installed", "deployment": {"watchdog": True}, "pending_stack_transaction": "/journal"},
106: {"status": "installed", "deployment": {"watchdog": True}},
}
for vmid, record in records.items():
(root / str(vmid)).mkdir()
(root / str(vmid) / "oci-compose.json").write_text(json.dumps({"vmid": vmid, **record}))
(root / "107").mkdir()
(root / "107" / "oci-compose.json").write_text("{broken")
self.assertEqual(watchdog.watched(root), {101: "Jellyfin", 106: "CT 106"})
class Start(unittest.TestCase):
def test_the_container_is_started_outside_the_service(self):
from unittest.mock import patch
with patch.object(watchdog.subprocess, "run") as run:
run.return_value.returncode = 0
self.assertTrue(watchdog.start(109))
run.return_value.returncode = 255
self.assertFalse(watchdog.start(109))
self.assertEqual(run.call_args.args[0], ["systemd-run", "--scope", "--quiet", "--collect", "pct", "start", "109"])
class Switch(unittest.TestCase):
"""Turning the watchdog on or off for a whole application."""
def registry(self, folder):
root = Path(folder)
ids = {101: "11111111-1111-4111-8111-111111111111", 102: "22222222-2222-4222-8222-222222222222",
103: "33333333-3333-4333-8333-333333333333"}
stack = {"members": [{"vmid": 101}, {"vmid": 102}]}
records = {101: {"stack": stack, "stack_member": {"primary_vmid": 101}},
102: {"stack_member": {"primary_vmid": 101}}, 103: {}}
for vmid, extra in records.items():
(root / str(vmid)).mkdir()
(root / str(vmid) / "oci-compose.json").write_text(json.dumps({
"schema_version": 1, "vmid": vmid, "installation_id": ids[vmid], "status": "installed",
"deployment": {"onboot": True}, **extra}))
return root
def flags(self, root):
return {int(path.parent.name): json.loads(path.read_text())["deployment"].get("watchdog")
for path in root.glob("*/oci-compose.json")}
def test_every_container_of_the_application_takes_the_choice_and_keeps_the_rest_of_its_record(self):
from unittest.mock import patch
import oci_carried_record
with tempfile.TemporaryDirectory() as folder:
root = self.registry(folder)
with patch.object(watchdog, "ensure_service") as service, \
patch.object(oci_carried_record, "carry", return_value="carried") as carry:
self.assertEqual(watchdog.set_watchdog(root, 102, True), [101, 102])
self.assertEqual(self.flags(root), {101: True, 102: True, 103: None})
service.assert_called_once_with()
self.assertEqual([call.args[1] for call in carry.call_args_list], [101, 102])
self.assertEqual(watchdog.set_watchdog(root, 103, True), [103])
self.assertEqual(watchdog.set_watchdog(root, 101, False), [101, 102])
self.assertEqual(self.flags(root), {101: False, 102: False, 103: True})
self.assertEqual(service.call_count, 2)
record = json.loads((root / "101" / "oci-compose.json").read_text())
self.assertEqual(record["deployment"]["onboot"], True)
self.assertEqual(record["stack"]["members"], [{"vmid": 101}, {"vmid": 102}])
if __name__ == "__main__":
unittest.main()
@@ -49,6 +49,11 @@
"The editor opens with the current contract; the CT is rebuilt with the changes",
"The data of container disks and host directories"
],
[
"Watchdog: restart the application when it crashes",
"Whether the application is started again when it stops on its own. Nothing is rebuilt",
"The whole installation; only the choice is saved in the contract"
],
[
"Remove: delete the application and its containers",
"The LXC, or every member of a stack, and their contracts are deleted",
@@ -58,7 +63,7 @@
}
},
{
"p": "For a multi-container application the menu offers <strong>Update every container of the application</strong>, <strong>Modify extra paths and devices</strong>, <strong>Recreate every container with its saved configuration</strong> and the removal. Modify adds or removes the extra paths and devices of the application container without rebuilding anything and, in Immich, changes what runs recognition. Recreate rebuilds every container from the image it was installed with, without looking for a newer one."
"p": "For a multi-container application the menu offers <strong>Update every container of the application</strong>, <strong>Modify extra paths and devices</strong>, <strong>Recreate every container with its saved configuration</strong>, the watchdog and the removal. Modify adds or removes the extra paths and devices of the application container without rebuilding anything and, in Immich, changes what runs recognition. Recreate rebuilds every container from the image it was installed with, without looking for a newer one."
},
{
"figure": {
@@ -158,6 +163,52 @@
}
]
},
{
"id": "watchdog",
"title": "Watchdog: restarting an application that stops",
"intro": "Proxmox has no restart policy: when the process of an application container ends, the container stays stopped. With the watchdog on, an application that stops on its own is started again. It does for an OCI application what <code>restart: unless-stopped</code> does in a Compose file.",
"blocks": [
{
"table": {
"headers": [
"What happens",
"With the watchdog on"
],
"rows": [
[
"The application crashes or its process ends",
"The container is started again within seconds"
],
[
"It is stopped or shut down from Proxmox, from the Monitor or with <code>pct</code>",
"It stays stopped until somebody starts it"
],
[
"A backup, a migration or a ProxMenux operation is working on it",
"Nothing is done until that ends"
],
[
"It was already stopped when the host started",
"It stays stopped; <strong>Start with Proxmox</strong> decides that"
],
[
"It keeps stopping right after starting",
"An application that stops again within seconds of starting is tried five times, waiting 30 seconds, 1, 2 and 5 minutes between them. Then it is left stopped and a notification is sent"
]
]
}
},
{
"p": "The installer asks it in both installation modes and proposes what the recipe of the application declares. It is changed later from <strong>Manage installed OCI applications</strong> or with the <strong>Watchdog</strong> switch of the container in <monitorLink>ProxMenux Monitor</monitorLink>, and it applies to every container of a multi-container application. One service of the host, <code>proxmenux-oci-watchdog.service</code>, watches every application that asked for it; its installation is listed in the Changes tab of Audit & Report."
},
{
"calloutInfo": {
"title": "What the host cannot tell",
"body": "The exit code of the application does not reach the host, so a clean exit and a crash look the same and both are restarted, as <code>always</code> and <code>unless-stopped</code> do in Docker. A stop asked outside Proxmox, with <code>lxc-stop</code> or <code>systemctl</code>, leaves no stop task and is taken as a stop of its own."
}
}
]
},
{
"id": "restore",
"title": "Backup, restore and another host",
@@ -25,6 +25,10 @@
"What changes for an OCI container"
],
"rows": [
[
"Status",
"A <strong>Watchdog</strong> switch under Start at boot: the application is started again when it stops on its own"
],
[
"App",
"The application is identified from the record, and updates are tracked by image"
@@ -49,6 +49,11 @@
"El editor se abre con el contrato actual; el CT se reconstruye con los cambios",
"Los datos de los discos del contenedor y de los directorios del host"
],
[
"Vigilancia: reiniciar la aplicación cuando falla",
"Si la aplicación se inicia de nuevo cuando se detiene sola. No se reconstruye nada",
"Toda la instalación; solo se guarda la elección en el contrato"
],
[
"Eliminar: la aplicación y sus contenedores",
"Se eliminan el LXC, o todos los miembros de una pila, y sus contratos",
@@ -58,7 +63,7 @@
}
},
{
"p": "Para una aplicación multicontenedor el menú ofrece <strong>Actualizar cada contenedor de la aplicación</strong>, <strong>Modificar rutas extra y dispositivos</strong>, <strong>Recrear todos los contenedores con su configuración guardada</strong> y la eliminación. Modificar añade o elimina las rutas y los dispositivos extra del contenedor de la aplicación sin reconstruir nada y, en Immich, cambia qué ejecuta el reconocimiento. Recrear reconstruye cada contenedor desde la imagen con la que se instaló, sin buscar una más nueva."
"p": "Para una aplicación multicontenedor el menú ofrece <strong>Actualizar cada contenedor de la aplicación</strong>, <strong>Modificar rutas extra y dispositivos</strong>, <strong>Recrear todos los contenedores con su configuración guardada</strong>, la vigilancia y la eliminación. Modificar añade o elimina las rutas y los dispositivos extra del contenedor de la aplicación sin reconstruir nada y, en Immich, cambia qué ejecuta el reconocimiento. Recrear reconstruye cada contenedor desde la imagen con la que se instaló, sin buscar una más nueva."
},
{
"figure": {
@@ -158,6 +163,52 @@
}
]
},
{
"id": "watchdog",
"title": "Vigilancia: reiniciar una aplicación que se detiene",
"intro": "Proxmox no tiene política de reinicio: cuando termina el proceso de un contenedor de aplicación, el contenedor se queda parado. Con la vigilancia activada, una aplicación que se detiene sola se inicia de nuevo. Hace por una aplicación OCI lo que <code>restart: unless-stopped</code> hace en un fichero Compose.",
"blocks": [
{
"table": {
"headers": [
"Qué ocurre",
"Con la vigilancia activada"
],
"rows": [
[
"La aplicación falla o su proceso termina",
"El contenedor se inicia de nuevo en segundos"
],
[
"Se detiene o se apaga desde Proxmox, desde el Monitor o con <code>pct</code>",
"Se queda parado hasta que alguien lo inicia"
],
[
"Un backup, una migración o una operación de ProxMenux está trabajando en él",
"No se hace nada hasta que termina"
],
[
"Ya estaba parado cuando arrancó el host",
"Se queda parado; eso lo decide <strong>Iniciar con Proxmox</strong>"
],
[
"No deja de detenerse nada más arrancar",
"Una aplicación que vuelve a detenerse a los pocos segundos de arrancar se intenta cinco veces, con esperas de 30 segundos, 1, 2 y 5 minutos entre ellas. Después se deja parada y se envía una notificación"
]
]
}
},
{
"p": "El instalador lo pregunta en los dos modos de instalación y propone lo que declara la receta de la aplicación. Después se cambia desde <strong>Gestionar aplicaciones OCI instaladas</strong> o con el interruptor <strong>Vigilancia</strong> del contenedor en <monitorLink>ProxMenux Monitor</monitorLink>, y se aplica a todos los contenedores de una aplicación multicontenedor. Un único servicio del host, <code>proxmenux-oci-watchdog.service</code>, vigila todas las aplicaciones que lo pidieron; su instalación aparece en la pestaña Cambios de Audit & Report."
},
{
"calloutInfo": {
"title": "Lo que el host no puede distinguir",
"body": "El código de salida de la aplicación no llega al host, así que una salida limpia y un fallo se ven igual y ambos se reinician, como hacen <code>always</code> y <code>unless-stopped</code> en Docker. Una parada pedida fuera de Proxmox, con <code>lxc-stop</code> o <code>systemctl</code>, no deja tarea de parada y se toma como una parada propia."
}
}
]
},
{
"id": "restore",
"title": "Backup, restauración y otro host",
@@ -25,6 +25,10 @@
"Qué cambia en un contenedor OCI"
],
"rows": [
[
"Estado",
"Un interruptor <strong>Vigilancia</strong> bajo Iniciar al arrancar: la aplicación se inicia de nuevo cuando se detiene sola"
],
[
"App",
"La aplicación se identifica desde el registro y sus actualizaciones se siguen por la imagen"