Merge pull request #420 from MacRimi/oci-restore-recovery-amd-gpu

OCI: recover applications after a Proxmox reinstall, cluster records and AMD GPU profiles
This commit is contained in:
MacRimi
2026-10-04 21:10:16 +02:00
committed by GitHub
71 changed files with 4555 additions and 128 deletions
@@ -41,6 +41,8 @@ class SelectionSetupWording(TestCase):
fn = extracted('_interactive_management', os=SimpleNamespace(geteuid=lambda: 0),
shutil=SimpleNamespace(which=lambda _: '/fake/pct'), saved_inventory=lambda _: [],
_clean_orphans=lambda project: None,
recover_automatically=lambda project: None, offer_recovery=lambda project, ui: None,
carry_records=lambda project, **options: {},
translate=lambda s: s)
fn(Path('/fixture'), ui)
ui.message.assert_called_once_with(EMPTY, 'OCI management')
@@ -130,6 +130,7 @@ class UpdateWording(unittest.TestCase):
scope = {'sys': SimpleNamespace(path=[], executable='python3'),
'translate': lambda text: text, 'check_selected': lambda project, row: row,
'images': SimpleNamespace(offer_removal=lambda *args: None),
'carry_records': lambda project, **options: {},
'_run_lifecycle': lambda command, title: events.append(('run', command, title)) or True}
row = {'vmid': 101, 'reason': 'matched', 'stack': False, 'pending': False, 'status': 'installed'}
decision = False
+27 -6
View File
@@ -51,6 +51,7 @@ interface OciInstanceInfo {
members: number[]
host_directories: boolean
pending: boolean
restored?: boolean
}
interface LxcUpdateCheck {
@@ -950,7 +951,10 @@ export function VirtualMachines() {
// Set when the open container was installed by OCI manager Apps: its
// Updates tab replaces the image instead of updating packages.
const [ociInstance, setOciInstance] = useState<OciInstanceInfo | null>(null)
const [ociAction, setOciAction] = useState<{ vmid: number; action: "update" | "recreate" } | null>(null)
// A container restored from a backup carries the mark of its installation
// and has no record on this host until it is recovered from the menu.
const [ociRestored, setOciRestored] = useState(false)
const [ociAction, setOciAction] = useState<{ vmid: number; action: "update" | "recreate" | "recover" } | null>(null)
// Firewall log state — fetched only when the operator opens that tab
// so a CT/VM without firewall use doesn't pay the pvesh cost on every
@@ -1260,6 +1264,7 @@ export function VirtualMachines() {
setFirewallEnabled(true)
setConsoleLogAvailable(false)
setOciInstance(null)
setOciRestored(false)
// Prime UI from last-known payloads so a reopened guest never
// flashes "Loading…" — the backend and this cache both revalidate
@@ -1429,10 +1434,12 @@ export function VirtualMachines() {
.then((r) => {
setConsoleLogAvailable(!!r?.console_log)
setOciInstance(r?.oci_instance ? r : null)
setOciRestored(!!r?.restored)
})
.catch(() => {
setConsoleLogAvailable(false)
setOciInstance(null)
setOciRestored(false)
})
// An operation that is still running when the modal opens clears on its
@@ -5550,7 +5557,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
{/* Branch 1 — OCI-image container not installed by
OCI manager Apps */}
{!selectedVM.update_check?.managed_oci_app && !ociInstance &&
selectedVM.update_check?.is_oci_lxc && (
(ociRestored || selectedVM.update_check?.is_oci_lxc) && (
<Card className="border border-border bg-card/50">
<CardContent className="p-4 space-y-2">
<div className="flex items-center gap-2 mb-1">
@@ -5558,12 +5565,24 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
<Container className="h-4 w-4 text-blue-400" />
</div>
<h3 className="text-sm font-semibold text-foreground">
{t("vmLxc.updates.ociTitle")}
{t(ociRestored ? "vmLxc.updates.ociRestoredTitle" : "vmLxc.updates.ociTitle")}
</h3>
</div>
<p className="text-sm text-muted-foreground leading-relaxed">
{t("vmLxc.updates.ociBody")}
{t(ociRestored ? "vmLxc.updates.ociRestoredBody" : "vmLxc.updates.ociBody")}
</p>
{ociRestored && (
<div className="mt-4 pt-4 border-t border-border/50 flex flex-wrap justify-end gap-2">
<Button
size="sm"
className="border border-input bg-background text-foreground/80 hover:bg-accent hover:text-accent-foreground"
onClick={() => setOciAction({ vmid: selectedVM.vmid, action: "recover" })}
>
<RotateCcw className="h-4 w-4 mr-1.5" />
{t("vmLxc.ociUpdates.recover")}
</Button>
</div>
)}
</CardContent>
</Card>
)}
@@ -5582,7 +5601,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
Individual actions stay with their section. The
optional reusable bulk action is configured in its
own card immediately before Options. */}
{!selectedVM.update_check?.managed_oci_app && !ociInstance &&
{!selectedVM.update_check?.managed_oci_app && !ociInstance && !ociRestored &&
!selectedVM.update_check?.is_oci_lxc && (() => {
const uc = selectedVM.update_check
const osUpdateStatusKnown = !!uc && !uc.error
@@ -7838,7 +7857,9 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
}}
scriptPath="/usr/local/share/proxmenux/scripts/oci/manage_instance.sh"
scriptName="oci_manage_instance"
title={ociAction.action === "update" ? t("vmLxc.ociUpdates.terminalTitleUpdate") : t("vmLxc.ociUpdates.terminalTitleRecreate")}
title={ociAction.action === "update" ? t("vmLxc.ociUpdates.terminalTitleUpdate")
: ociAction.action === "recover" ? t("vmLxc.ociUpdates.terminalTitleRecover")
: t("vmLxc.ociUpdates.terminalTitleRecreate")}
description={t("vmLxc.ociUpdates.terminalDescription")}
params={{
VMID: String(ociAction.vmid),
+3
View File
@@ -1204,6 +1204,7 @@
"recover": "Wiederherstellen",
"terminalTitleUpdate": "OCI-Anwendung aktualisieren",
"terminalTitleRecreate": "OCI-Anwendung neu erstellen",
"terminalTitleRecover": "Wiederhergestellte OCI-Anwendung registrieren",
"terminalDescription": "Derselbe Ablauf wie OCI manager Apps → Installierte OCI-Anwendungen verwalten.",
"keepBackup": "Das vor dem Update erstellte Backup behalten",
"noKeepBackup": "Das vor dem Update erstellte Backup wird nicht behalten",
@@ -1415,6 +1416,8 @@
"noManagedUpdateInfo": "Noch keine Update-Informationen – prüfen Sie unter Sicherheit → Secure Gateway.",
"ociTitle": "OCI-Image-Container",
"ociBody": "Dieser Container wurde aus einem OCI-Image (Docker) erstellt. Die Updateverwaltung für OCI-Container wird mit der kommenden OCI-Installationsfunktion geliefert – Updates erstellen den Container anhand eines neueren Image-Tags neu, anstatt Pakete darin zu patchen.",
"ociRestoredTitle": "Wiederhergestellte OCI-Anwendung",
"ociRestoredBody": "Dieser Container wurde aus einem Backup wiederhergestellt und dieser Host hat keinen Eintrag zu seiner Anwendung, daher kann er von hier aus noch nicht aktualisiert werden. Wiederherstellen registriert ihn erneut auf diesem Host, genauso wie OCI manager Apps → Installierte OCI-Anwendungen verwalten.",
"dockerImagesTitle": "Docker-Images",
"dockerAppTitle": "Docker",
"dockerImagesSubheading": "Images",
+3
View File
@@ -1225,6 +1225,7 @@
"recover": "Recover",
"terminalTitleUpdate": "Update OCI application",
"terminalTitleRecreate": "Recreate OCI application",
"terminalTitleRecover": "Recover restored OCI application",
"terminalDescription": "The same flow as OCI manager Apps → Manage installed OCI applications.",
"keepBackup": "Keep the backup taken before updating",
"noKeepBackup": "The backup taken before updating is not kept",
@@ -1436,6 +1437,8 @@
"noManagedUpdateInfo": "No update information yet — check from Security → Secure Gateway.",
"ociTitle": "OCI image container",
"ociBody": "This container was created from an OCI image outside OCI manager Apps. It is updated by replacing its image, which ProxMenux does not manage for it.",
"ociRestoredTitle": "Restored OCI application",
"ociRestoredBody": "This container was restored from a backup and this host has no record of its application, so it cannot be updated from here yet. Recover registers it again on this host, the same as OCI manager Apps → Manage installed OCI applications.",
"dockerImagesTitle": "Docker images",
"dockerAppTitle": "Docker",
"dockerImagesSubheading": "Images",
+3
View File
@@ -1204,6 +1204,7 @@
"recover": "Recuperar",
"terminalTitleUpdate": "Actualizar aplicación OCI",
"terminalTitleRecreate": "Recrear aplicación OCI",
"terminalTitleRecover": "Recuperar aplicación OCI restaurada",
"terminalDescription": "El mismo recorrido que OCI manager Apps → Gestionar aplicaciones OCI instaladas.",
"keepBackup": "Conservar el backup realizado antes de actualizar",
"noKeepBackup": "No se conserva el backup realizado antes de actualizar",
@@ -1415,6 +1416,8 @@
"noManagedUpdateInfo": "Aún no hay información de actualización: verifique desde Seguridad → Secure Gateway.",
"ociTitle": "Contenedor de imágenes OCI",
"ociBody": "Este contenedor se creó desde una imagen OCI fuera de OCI manager Apps. Se actualiza sustituyendo su imagen, algo que ProxMenux no gestiona en este caso.",
"ociRestoredTitle": "Aplicación OCI restaurada",
"ociRestoredBody": "Este contenedor se restauró desde un backup y este host no tiene el registro de su aplicación, así que todavía no se puede actualizar desde aquí. Recuperar lo registra de nuevo en este host, igual que OCI manager Apps → Gestionar aplicaciones OCI instaladas.",
"dockerImagesTitle": "Imágenes Docker",
"dockerAppTitle": "Docker",
"dockerImagesSubheading": "Imágenes",
+3
View File
@@ -1204,6 +1204,7 @@
"recover": "Récupérer",
"terminalTitleUpdate": "Mettre à jour l'application OCI",
"terminalTitleRecreate": "Recréer l'application OCI",
"terminalTitleRecover": "Récupérer l'application OCI restaurée",
"terminalDescription": "Le même parcours que OCI manager Apps → Gérer les applications OCI installées.",
"keepBackup": "Conserver la sauvegarde prise avant la mise à jour",
"noKeepBackup": "La sauvegarde prise avant la mise à jour n'est pas conservée",
@@ -1415,6 +1416,8 @@
"noManagedUpdateInfo": "Aucune information de mise à jour pour l'instant - vérifiez depuis Sécurité → Secure Gateway.",
"ociTitle": "Conteneur d'images OCI",
"ociBody": "Ce conteneur a été créé à partir d'une image OCI (Docker). La gestion des mises à jour pour les conteneurs OCI est fournie avec la prochaine fonctionnalité d'installation OCI : les mises à jour reconstruiront le conteneur à partir d'une balise d'image plus récente plutôt que de corriger les packages à l'intérieur.",
"ociRestoredTitle": "Application OCI restaurée",
"ociRestoredBody": "Ce conteneur a été restauré depuis une sauvegarde et cet hôte n'a pas l'enregistrement de son application ; il ne peut donc pas encore être mis à jour d'ici. Récupérer l'enregistre à nouveau sur cet hôte, comme OCI manager Apps → Gérer les applications OCI installées.",
"dockerImagesTitle": "Images Docker",
"dockerAppTitle": "Docker",
"dockerImagesSubheading": "Images",
+3
View File
@@ -1204,6 +1204,7 @@
"recover": "Recupera",
"terminalTitleUpdate": "Aggiorna applicazione OCI",
"terminalTitleRecreate": "Ricrea applicazione OCI",
"terminalTitleRecover": "Recupera applicazione OCI ripristinata",
"terminalDescription": "Lo stesso percorso di OCI manager Apps → Gestisci le applicazioni OCI installate.",
"keepBackup": "Conserva il backup eseguito prima dell'aggiornamento",
"noKeepBackup": "Il backup eseguito prima dell'aggiornamento non viene conservato",
@@ -1415,6 +1416,8 @@
"noManagedUpdateInfo": "Nessuna informazione di aggiornamento ancora: controlla da Sicurezza → Secure Gateway.",
"ociTitle": "Contenitore di immagini OCI",
"ociBody": "Questo contenitore è stato creato da un'immagine OCI (Docker). La gestione degli aggiornamenti per i contenitori OCI arriverà con la prossima funzionalità di installazione OCI: gli aggiornamenti ricostruiranno il contenitore da un tag immagine più recente anziché applicare patch ai pacchetti all'interno.",
"ociRestoredTitle": "Applicazione OCI ripristinata",
"ociRestoredBody": "Questo contenitore è stato ripristinato da un backup e questo host non ha il registro della sua applicazione, quindi non può ancora essere aggiornato da qui. Recupera lo registra di nuovo su questo host, come OCI manager Apps → Gestisci le applicazioni OCI installate.",
"dockerImagesTitle": "Immagini Docker",
"dockerAppTitle": "Docker",
"dockerImagesSubheading": "Immagini",
+3
View File
@@ -1204,6 +1204,7 @@
"recover": "Recuperar",
"terminalTitleUpdate": "Atualizar aplicação OCI",
"terminalTitleRecreate": "Recriar aplicação OCI",
"terminalTitleRecover": "Recuperar aplicação OCI restaurada",
"terminalDescription": "O mesmo percurso de OCI manager Apps → Gerir aplicações OCI instaladas.",
"keepBackup": "Manter o backup feito antes da atualização",
"noKeepBackup": "O backup feito antes da atualização não é mantido",
@@ -1415,6 +1416,8 @@
"noManagedUpdateInfo": "Nenhuma informação de atualização ainda — verifique em Segurança → Secure Gateway.",
"ociTitle": "Contêiner de imagem OCI",
"ociBody": "Este contêiner foi criado a partir de uma imagem OCI (Docker). O gerenciamento de atualizações para contêineres OCI vem com o próximo recurso de instalação do OCI – as atualizações reconstruirão o contêiner a partir de uma tag de imagem mais recente, em vez de corrigir os pacotes internos.",
"ociRestoredTitle": "Aplicação OCI restaurada",
"ociRestoredBody": "Este contêiner foi restaurado a partir de um backup e este host não tem o registro da sua aplicação, por isso ainda não pode ser atualizado daqui. Recuperar o registra novamente neste host, tal como OCI manager Apps → Gerenciar aplicações OCI instaladas.",
"dockerImagesTitle": "Imagens Docker",
"dockerAppTitle": "Docker",
"dockerImagesSubheading": "Imagens",
+3
View File
@@ -1225,6 +1225,7 @@
"recover": "Obnoviť",
"terminalTitleUpdate": "Aktualizovať OCI aplikáciu",
"terminalTitleRecreate": "Znovu vytvoriť OCI aplikáciu",
"terminalTitleRecover": "Obnoviť obnovenú OCI aplikáciu",
"terminalDescription": "Použije sa rovnaký postup ako v ponuke Správca OCI aplikácií → Spravovať nainštalované OCI aplikácie.",
"keepBackup": "Ponechať zálohu vytvorenú pred aktualizáciou",
"noKeepBackup": "Záloha vytvorená pred aktualizáciou sa neponecháva",
@@ -1436,6 +1437,8 @@
"noManagedUpdateInfo": "Zatiaľ nie sú žiadne informácie o aktualizácii – skontrolujte ich v časti Zabezpečenie → Secure Gateway.",
"ociTitle": "Kontajner z OCI obrazu",
"ociBody": "Tento kontajner bol vytvorený z OCI (Docker) obrazu mimo Správcu OCI aplikácií. Aktualizuje sa nahradením obrazu, čo ProxMenux pri tomto kontajneri nespravuje.",
"ociRestoredTitle": "Obnovená OCI aplikácia",
"ociRestoredBody": "Tento kontajner bol obnovený zo zálohy a tento hostiteľ nemá záznam o jeho aplikácii, preto ho zatiaľ nemožno aktualizovať odtiaľto. Obnoviť ho znova zaregistruje na tomto hostiteľovi, rovnako ako OCI manager Apps → Spravovať nainštalované OCI aplikácie.",
"dockerImagesTitle": "Docker obrazy",
"dockerAppTitle": "Docker",
"dockerImagesSubheading": "Obrazy",
+3
View File
@@ -1204,6 +1204,7 @@
"recover": "Återställ",
"terminalTitleUpdate": "Uppdatera OCI-applikation",
"terminalTitleRecreate": "Återskapa OCI-applikation",
"terminalTitleRecover": "Återställ återställd OCI-applikation",
"terminalDescription": "Samma flöde som OCI manager Apps → Hantera installerade OCI-applikationer.",
"keepBackup": "Behåll säkerhetskopian som tas före uppdateringen",
"noKeepBackup": "Säkerhetskopian som tas före uppdateringen behålls inte",
@@ -1415,6 +1416,8 @@
"noManagedUpdateInfo": "Ingen uppdateringsinformation ännu — kolla från Säkerhet → Secure Gateway.",
"ociTitle": "OCI-bildbehållare",
"ociBody": "Den här behållaren skapades från en OCI-bild (Docker). Uppdateringshantering för OCI-behållare kommer med den kommande OCI-installationsfunktionen - uppdateringar kommer att bygga om behållaren från en nyare bildtagg snarare än att patcha paket inuti.",
"ociRestoredTitle": "Återställd OCI-applikation",
"ociRestoredBody": "Den här behållaren återställdes från en säkerhetskopia och den här värden har ingen post för dess applikation, så den kan ännu inte uppdateras härifrån. Återställ registrerar den igen på den här värden, på samma sätt som OCI manager Apps → Hantera installerade OCI-applikationer.",
"dockerImagesTitle": "Docker-avbilder",
"dockerAppTitle": "Docker",
"dockerImagesSubheading": "Avbilder",
+12 -2
View File
@@ -4347,6 +4347,7 @@ def check_app(
"error": result.get("error"),
"checked_at": _now_iso(),
"installed_digest": result.get("installed_digest"),
"installed_registry_digest": result.get("installed_registry_digest"),
"latest_digest": result.get("latest_digest"),
"image_created": result.get("image_created"),
"latest_image_created": result.get("latest_image_created"),
@@ -5598,6 +5599,7 @@ def _oci_instance_meta(vmid) -> Optional[dict]:
contract = template.get("container_contract") or {}
image = contract.get("image") or {}
observed_image = (record.get("observed") or {}).get("image") or {}
registry_digest = str((record.get("observed") or {}).get("resolved_registry_digest") or "").strip()
# A stack member carries only its own image contract; the presentation
# belongs to the stack it is part of, which records its title, site,
# category and the endpoint the stack is reached on.
@@ -5702,6 +5704,9 @@ def _oci_instance_meta(vmid) -> Optional[dict]:
# The exact image this container was created from. Its digest is what
# an update is decided on; the version label is only for reading.
"installed_digest": str(observed_image.get("manifest_digest") or "").strip() or None,
# The digest the registry served that image under. An image published
# with a Docker-format manifest is saved under another one.
"registry_digest": registry_digest if re.fullmatch(r"sha256:[0-9a-f]{64}", registry_digest) else None,
"architecture": str(observed_image.get("architecture") or "").strip() or None,
}
@@ -5789,16 +5794,21 @@ def _oci_image_versions(vmid, known: Optional[dict] = None, with_latest: bool =
return {**result, "error": f"OCI engine unavailable: {exc}"}
known = known or {}
# Deciding an update needs the digest the registry gave the installed image.
if (known.get("installed_digest") == installed_digest
and (known.get("installed_registry_digest") or not with_latest)
and (known.get("installed_version") or known.get("image_created"))):
result["installed_version"] = known.get("installed_version")
result["image_created"] = known.get("image_created")
result["installed_registry_digest"] = known.get("installed_registry_digest")
else:
try:
installed = _oci_resolve(
module, f"{_oci_repository(reference)}@{installed_digest}", architecture)
module, f"{_oci_repository(reference)}@{meta.get('registry_digest') or installed_digest}",
architecture)
result["installed_version"] = installed.get("version")
result["image_created"] = installed.get("created")
result["installed_registry_digest"] = installed.get("manifest_digest")
except Exception as exc:
return {**result, "error": f"could not read the installed image: {exc}"}
if not result.get("installed_version"):
@@ -5815,7 +5825,7 @@ def _oci_image_versions(vmid, known: Optional[dict] = None, with_latest: bool =
# The image decides. An application whose version did not move can still
# have a new image — a rebuild on a patched base — and that is an update
# for a container whose application only changes when its image does.
replaced = bool(latest_digest) and latest_digest != installed_digest
replaced = bool(latest_digest) and latest_digest != (result.get("installed_registry_digest") or installed_digest)
result.update(latest_digest=latest_digest, latest_version=latest.get("version"),
latest_image_created=latest.get("created"), update_available=replaced)
return result
+34 -1
View File
@@ -3,13 +3,16 @@
Read-only view of the installation record OCI manager Apps keeps for every
container it created: whether the container is one, whether it belongs to a
multi-container application, whether it uses host directories (which its
backup does not revert) and whether an operation is pending. Nothing here
backup does not revert), whether an operation is pending and whether it was
restored from a backup and is not registered on this host yet. Nothing here
changes the record or runs inside the container.
"""
from __future__ import annotations
import json
import os
import re
import socket
import oci_console_logs
@@ -26,6 +29,30 @@ def _record(vmid: int) -> dict | None:
return record if isinstance(record, dict) else None
def _installation(vmid: int) -> str | None:
"""The installation the container says it belongs to: the mark OCI manager
Apps leaves in its notes, which a backup keeps."""
path = f"/etc/pve/nodes/{socket.gethostname().split('.', 1)[0]}/lxc/{int(vmid)}.conf"
try:
with open(path, encoding="utf-8", errors="ignore") as handle:
text = handle.read().split("\n[", 1)[0]
except OSError:
return None
match = re.search(r"^#.*proxmenux-instance=([0-9a-f-]{36})(?![0-9a-f-])", text, re.MULTILINE)
return match.group(1) if match else None
def _unrecoverable(vmid: int, installation: str) -> bool:
"""A restored container that carries no copy of its record: the menu
found nothing to recover it from and left it as an ordinary container."""
try:
with open(os.path.join(ROOT, ".unrecoverable.json"), encoding="utf-8") as handle:
value = json.load(handle)
except (OSError, ValueError):
return False
return isinstance(value, dict) and value.get(str(int(vmid))) == installation
def _host_dirs(record: dict) -> bool:
mounts = (record.get("deployment") or {}).get("mounts") or []
return any(isinstance(m, dict) and m.get("type") == "host-bind" for m in mounts)
@@ -42,8 +69,14 @@ def info(vmid: int) -> dict:
"members": [],
"host_directories": False,
"pending": False,
"restored": False,
}
record = _record(vmid)
installation = _installation(vmid)
if installation and (record is None or record.get("installation_id") != installation):
# Restored from a backup: the record stayed on the host it came from.
result["restored"] = not _unrecoverable(vmid, installation)
return result
if record is None:
return result
primary_id = int((record.get("stack_member") or {}).get("primary_vmid") or vmid)
@@ -0,0 +1,76 @@
"""An image published with a Docker-format manifest is saved for Proxmox
under another digest than the one its registry serves. Its installed version
and its updates are read through the digest the registry knows."""
import sys
from pathlib import Path
import unittest
from unittest.mock import patch
SCRIPTS = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(SCRIPTS))
import lxc_apps
ARCHIVE = "sha256:" + "a1" * 32
INDEX = "sha256:" + "b2" * 32
PLATFORM = "sha256:" + "c3" * 32
NEWER = "sha256:" + "d4" * 32
class Registry:
"""A registry that knows the index and the platform manifest, never the archive."""
def __init__(self, tag=PLATFORM):
self.tag, self.asked = tag, []
def resolve_candidate(self, reference, architecture):
self.asked.append(reference)
if reference.endswith("@" + ARCHIVE):
raise RuntimeError("manifest unknown")
digest = self.tag if "@" not in reference else PLATFORM
return {"manifest_digest": digest, "version": "12.1" if digest == PLATFORM else "12.2",
"created": "2026-09-15T01:13:55Z"}
class DockerManifestVersionTests(unittest.TestCase):
def versions(self, registry, registry_digest=INDEX, known=None):
meta = {"image_reference": "jellyfin/jellyfin:latest", "installed_digest": ARCHIVE,
"registry_digest": registry_digest, "architecture": "amd64", "repository": None}
with patch.object(lxc_apps, "_oci_operation_running", return_value=False), \
patch.object(lxc_apps, "_oci_instance_meta", return_value=meta), \
patch.object(lxc_apps, "_oci_state_module", return_value=registry), \
patch.object(lxc_apps.time, "sleep"):
return lxc_apps._oci_image_versions(105, known=known)
def test_the_installed_image_is_read_by_the_digest_the_registry_served(self):
registry = Registry()
result = self.versions(registry)
self.assertNotIn("error", result)
self.assertEqual(registry.asked[0], "jellyfin/jellyfin@" + INDEX)
self.assertEqual((result["installed_version"], result["installed_registry_digest"]), ("12.1", PLATFORM))
self.assertFalse(result["update_available"])
self.assertEqual(result["installed_digest"], ARCHIVE)
def test_a_new_image_under_the_tag_is_an_update(self):
result = self.versions(Registry(tag=NEWER))
self.assertTrue(result["update_available"])
self.assertEqual((result["latest_digest"], result["latest_version"]), (NEWER, "12.2"))
def test_a_previous_answer_is_reused_with_its_registry_digest(self):
registry = Registry()
known = {"installed_digest": ARCHIVE, "installed_registry_digest": PLATFORM, "installed_version": "12.1",
"image_created": "2026-09-15T01:13:55Z"}
result = self.versions(registry, known=known)
self.assertEqual(registry.asked, ["jellyfin/jellyfin:latest"])
self.assertFalse(result["update_available"])
# An answer saved without the registry digest is asked again.
registry = Registry()
self.versions(registry, known={key: value for key, value in known.items() if key != "installed_registry_digest"})
self.assertEqual(registry.asked[0], "jellyfin/jellyfin@" + INDEX)
def test_a_record_without_the_registry_digest_is_read_by_its_own(self):
result = self.versions(Registry(), registry_digest=None)
self.assertIn("could not read the installed image", result["error"])
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,76 @@
"""A container restored from a backup carries the mark of its installation and
has no record on this host: the modal says so instead of treating it as an
ordinary container."""
import json
import sys
import tempfile
from pathlib import Path
import unittest
from unittest.mock import patch
SCRIPTS = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(SCRIPTS))
import oci_instance_info
APP = "83e373f2-5ccc-4548-9b89-22d0956d1c77"
OTHER = "488ed3cb-1145-477a-b90f-c46777d69fb2"
class RestoredInstanceTests(unittest.TestCase):
def setUp(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
self.root = Path(tmp.name)
for target, value in (("ROOT", str(self.root)),):
patcher = patch.object(oci_instance_info, target, value)
patcher.start()
self.addCleanup(patcher.stop)
patcher = patch.object(oci_instance_info.oci_console_logs, "configured_log", return_value=None)
patcher.start()
self.addCleanup(patcher.stop)
def record(self, vmid, identity):
folder = self.root / str(vmid)
folder.mkdir()
(folder / "oci-compose.json").write_text(json.dumps(
{"vmid": vmid, "installation_id": identity, "status": "installed", "deployment": {}}))
def info(self, vmid, installation):
with patch.object(oci_instance_info, "_installation", return_value=installation):
return oci_instance_info.info(vmid)
def test_a_marked_container_without_record_was_restored(self):
result = self.info(119, APP)
self.assertTrue(result["restored"])
self.assertFalse(result["oci_instance"])
def test_the_record_of_another_installation_does_not_register_it(self):
self.record(119, OTHER)
result = self.info(119, APP)
self.assertTrue(result["restored"])
self.assertFalse(result["oci_instance"])
def test_a_container_the_menu_could_not_recover_is_an_ordinary_one(self):
(self.root / ".unrecoverable.json").write_text(json.dumps({"119": APP}))
self.assertFalse(self.info(119, APP)["restored"])
self.assertTrue(self.info(119, OTHER)["restored"])
def test_a_registered_container_is_an_oci_instance(self):
self.record(119, APP)
result = self.info(119, APP)
self.assertFalse(result["restored"])
self.assertTrue(result["oci_instance"])
def test_an_ordinary_container_is_neither(self):
result = self.info(100, None)
self.assertFalse(result["restored"])
self.assertFalse(result["oci_instance"])
def test_the_mark_is_read_from_the_notes_of_the_container(self):
text = f"#<div>Tandoor</div>\n#<!-- proxmenux-instance={APP} -->\narch: amd64\n[snap]\n#proxmenux-instance={OTHER}\n"
with patch("builtins.open", unittest.mock.mock_open(read_data=text)):
self.assertEqual(oci_instance_info._installation(119), APP)
if __name__ == "__main__":
unittest.main()
+89 -2
View File
@@ -47,6 +47,7 @@
"A complete restore will:": "Una restauración completa:",
"A concurrent change was detected; the container is not removed": "Se detectó un cambio simultáneo; el contenedor no se retira",
"A container mount has a source, backup or permission different from the saved record": "Un contenedor de montaje tiene una fuente, backup o permiso diferente del registro guardado",
"A container without its record stays as an ordinary container and is not offered again.": "Un contenedor sin su registro queda como un contenedor normal y no se vuelve a ofrecer.",
"A coordinated operation has a saved journal. Continuing attempts to recover the previous stack where needed, or finish cleanup for a completed operation. Recovery or cleanup can fail.": "Una operación coordinada tiene un registro de operación guardado. Al continuar se intenta recuperar la pila anterior donde haga falta, o terminar la limpieza de una operación ya completada. La recuperación o la limpieza pueden fallar.",
"A coordinated operation is pending. The whole previous stack will be recovered, not only the selected member. If the operation already finished, the cleanup of its markers is completed.": "Está pendiente una operación coordinada. Toda la pila anterior se recuperará, no sólo el miembro seleccionado. Si la operación ya terminó, se completa la limpieza de sus marcadores.",
"A different host monitor include already exists; it is not overwritten:": "Un monitor de host diferente incluye ya existe; no está sobrescrito:",
@@ -61,6 +62,7 @@
"A host mount is not part of the journal; recovery blocked": "Un montaje del host no figura en el diario de la operación; recuperación bloqueada",
"A host reboot is required after this change.": "Es necesario reiniciar el host después de este cambio.",
"A host reboot is required before starting the VM. Reboot now?": "Es necesario reiniciar el host antes de iniciar la VM. ¿Reiniciar ahora?",
"A host script is not written over a symbolic link:": "No se escribe un script del host sobre un enlace simbólico:",
"A job with this ID already exists.": "Ya existe un trabajo con este ID.",
"A journal already exists; review or recover it before trying again": "Ya existe un diario de operación; revísalo o recupéralo antes de volver a intentarlo",
"A keyfile is installed at:": "un archivo de claves está instalado en:",
@@ -241,6 +243,7 @@
"Add latest Ceph support": "Añadir soporte para la última versión de Ceph",
"Add new PVE 9 enterprise repository (deb822 format) (Only if using enterprise):": "Agregue un nuevo repositorio empresarial PVE 9 (formato deb822) (solo si usa Enterprise):",
"Add or change a device": "Agregar o cambiar un dispositivo",
"Add or remove extra paths and devices": "Añadir o eliminar rutas y dispositivos extra",
"Add physical disk to VM via": "Agregue el disco físico a la VM mediante",
"Add share block in /etc/samba/smb.conf:": "Agregue el bloque para compartir en /etc/samba/smb.conf:",
"Add unprivileged flag to container configuration:": "Agregar la opción de contenedor sin privilegios a su configuración:",
@@ -380,6 +383,7 @@
"Another stack operation is pending": "Hay otra operación de pila pendiente",
"Application": "Aplicación",
"Application Specific": "Específico de la aplicación",
"Application registered on this host:": "Aplicación registrada en este host:",
"Application responding:": "Solicitud de respuesta:",
"Application responding; checking its stability...": "Aplicación respondiendo; comprobando su estabilidad...",
"Application suite: one independent LXC per selected application": "Suite de aplicación: un LXC independiente por aplicación seleccionada",
@@ -668,6 +672,7 @@
"CRITICAL: The selected disk is referenced by a RUNNING VM or CT.": "CRÍTICO: El disco seleccionado tiene referencia a una VM o CT EN EJECUCIÓN.",
"CT": "LXC",
"CT started successfully.": "El LXC se inició con éxito.",
"CT {vmid} is a host monitor and had a rule in the host firewall. Allow TCP port {port} from {subnet} through the firewall of this host? Existing firewall rules are not changed.": "El CT {vmid} es un monitor del host y tenía una regla en el firewall del host. ¿Permitir el puerto TCP {port} desde {subnet} en el firewall de este host? Las reglas existentes no se modifican.",
"CUDA requires a working NVIDIA driver": "CUDA requires un controlador NVIDIA de trabajo",
"CUDA requires the NVIDIA Container Toolkit on the host": "CUDA requiere el NVIDIA Container Toolkit en el host",
"Calculate all kinds of statistics from your (local) Emby or Jellyfin server": "Calcular todo tipo de estadísticas de su servidor (local) Emby o Jellyfin",
@@ -734,6 +739,8 @@
"Change it after the first login.": "Cámbialo después del primer inicio de sesión.",
"Change or add an environment variable?": "¿Cambiar o añadir una variable de entorno?",
"Change the access network?": "¿Cambiar la red de acceso?",
"Change the recognition now?": "¿Cambiar el reconocimiento ahora?",
"Change what runs recognition: CPU or GPU": "Cambiar qué ejecuta el reconocimiento: CPU o GPU",
"Changedetection.io provides free, open-source web page monitoring, notification and change detection.": "Changedetection.io proporciona monitorización gratuita de páginas web de código abierto, notificación y detección de cambios.",
"Changes applied. A system reboot is recommended for them to take full effect.": "Se aplicaron cambios. Se recomienda reiniciar el sistema para que surtan efecto completo.",
"Changes have been applied to the configuration file.": "Se han aplicado cambios al archivo de configuración.",
@@ -829,6 +836,7 @@
"Choose an action:": "Elige una acción:",
"Choose an export to mount:": "Elige una exportación para montar:",
"Choose an option:": "Elige una opción:",
"Choose another bridge for the container in Proxmox (Network) and run the recovery again.": "Elige otro bridge para el contenedor en Proxmox (Red) y vuelve a lanzar la recuperación.",
"Choose any host path you want to share with CTs.": "Elige cualquier ruta de host que desee compartir con los CT.",
"Choose conflict policy": "Elige una política de conflictos",
"Choose conflict policy for the source VM:": "Elige la política de conflictos para la máquina virtual de origen:",
@@ -1063,6 +1071,7 @@
"Conflicting path included in backup:": "Ruta conflictiva incluida en la copia de seguridad:",
"Conflicting utilities removed": "Se eliminaron las utilidades conflictivas",
"Connect a Coral Accelerator and try again.": "Conecte un Coral Accelerator y vuelva a intentarlo.",
"Connect it, or remove it from the container in Proxmox (Resources), and run the recovery again.": "Conéctalo, o elimínalo del contenedor en Proxmox (Recursos), y vuelve a lanzar la recuperación.",
"Connect your devices and users together in your own secure virtual private network.": "Conecta tus dispositivos y usuarios juntos en tu propia red privada virtual segura.",
"Connect your devices into a secure WireGuard®-based overlay network with SSO, MFA and granular access controls.": "Conecte sus dispositivos en una red de superposición segura de WireGuard® con SSO, MFA y controles de acceso granular.",
"Connected": "Conectado",
@@ -1404,6 +1413,7 @@
"Creating the container: preparing its disk...": "Creando el contenedor: preparando su disco...",
"Creating the container: reading the image...": "Creando el contenedor: leyendo la imagen...",
"Creating the initial administrator...": "Crear el administrador inicial...",
"Creating the private network...": "Creando la red privada...",
"Creating the temporary data container": "Creando el contenedor temporal para los datos",
"Creative & Design": "Creatividad y diseño",
"Credentials are correct": "Las credenciales son correctas",
@@ -2036,7 +2046,7 @@
"Every fail2ban jail ships disabled. Enable the ones you need in /config/fail2ban/jail.local, taking the ready-made jails in /config/fail2ban/jail.d/ as reference, then restart the container.": "Cada nave de prisión de fail2ban discapacitados. Habilitar los que necesites en /config/fail2ban/jail.local, tomando las cárceles listas en /config/fail2ban/jail.d/ como referencia, luego reiniciar el contenedor.",
"Every hour": "Cada hora",
"Every member of the stack is back to its previous installation.": "Cada miembro de la pila está de vuelta a su instalación anterior.",
"Every node of this cluster already has this file. If you restore this container on any other Proxmox host, copy it to the same path first, because the container backup does not include it:": "Todos los nodos de este clúster ya tienen este archivo. Si restauras este contenedor en cualquier otro host Proxmox, copia antes el archivo a la misma ruta, porque el backup del contenedor no lo incluye:",
"Every node of this cluster already has this file. A backup of the container does not include it: on another Proxmox host it is written again when the application is recovered from Manage installed OCI applications:": "Todos los nodos de este clúster ya tienen este archivo. Un backup del contenedor no lo incluye: en otro host Proxmox se escribe de nuevo al recuperar la aplicación desde Gestionar aplicaciones OCI instaladas:",
"Every path in this backup is kernel-tied: the restore applies these paths automatically via the safe-subset filter and re-merges the operator's tuning.": "Cada ruta en esta copia de seguridad está vinculada al kernel: la restauración aplica estas rutas automáticamente a través del filtro de subconjunto seguro y vuelve a fusionar el ajuste del operador.",
"Everything restorable in this backup will be restored": "Todo lo que se pueda restaurar en esta copia de seguridad será restaurado",
"Exact name of the remote": "Nombre exacto del remoto",
@@ -2751,6 +2761,7 @@
"If config issues occur:": "Si ocurren problemas de configuración:",
"If container won't start:": "Si el contenedor no se inicia:",
"If imported from ESXi: install qemu-guest-agent inside the guest OS": "Si se importa desde ESXi: instale qemu-guest-agent dentro del sistema operativo invitado",
"If it does not work well, the acceleration can be changed back from Manage installed OCI applications, without reinstalling.": "Si no funciona bien, la aceleración se puede cambiar de nuevo desde Gestionar aplicaciones OCI instaladas, sin reinstalar.",
"If it still fails, the NFS server export options must be changed on the server.": "Si aún falla, se deben cambiar las opciones de exportación del servidor NFS en el servidor.",
"If it warns about 'systemd-boot' meta-package, remove it:": "Si advierte sobre el metapaquete 'systemd-boot', elimínelo:",
"If mount fails (LVM):": "Si el montaje falla (LVM):",
@@ -3089,6 +3100,7 @@
"Invalid Python module:": "Módulo Python inválido:",
"Invalid Python package:": "Paquete de pitón inválido:",
"Invalid Python path in the repair:": "Camino de pitón inválido en la reparación:",
"Invalid ROCm generation override": "Generación forzada de ROCm no válida",
"Invalid Unpackerr variable": "Inválido Unpackerr variable",
"Invalid VFS cache mode": "Modo de caché VFS inválido",
"Invalid VMID": "VMID no válido",
@@ -3216,6 +3228,8 @@
"It works through the Docker engine of the host, and a native OCI container does not have one.": "Funciona a través del motor Docker del host, y un contenedor OCI nativo no tiene uno.",
"Italian": "italiano",
"Its Compose file asks for privileged mode; the container is created unprivileged and that mode is only offered as an option.": "Su archivo Compose pide un modo privilegiado; el contenedor se crea sin privilegios y ese modo sólo se ofrece como una opción.",
"Its ROCm profile is not offered.": "Su perfil ROCm no se ofrece.",
"Its addresses are fixed and are not changed.": "Sus direcciones son fijas y no se cambian.",
"Its final cleanup did not complete. Select the stack again in the OCI management menu to complete it.": "Su limpieza final no terminó. Seleccione la pila de nuevo en el menú de gestión OCI para completarla.",
"Its labels are not applied: they are read by other Docker tools.": "Sus etiquetas no se aplican: son leídas por otras herramientas Docker.",
"JC Channel logo applied": "Logotipo de JC Channel aplicado",
@@ -3451,6 +3465,7 @@
"MOTD configuration was already up to date": "la configuración de MOTD ya estaba actualizada",
"Machine Type": "Tipo de máquina",
"Machine learning": "Aprendizaje automático",
"Machine learning container prepared": "Contenedor Machine learning preparado",
"Machine learning profile not implemented; it is not replaced by CPU:": "Perfil de Machine learning no implementado; no se sustituye por CPU:",
"Machine type: q35": "Tipo de máquina: q35",
"Machine: q35": "Máquina: q35",
@@ -3590,6 +3605,7 @@
"Mount name": "Nombre del montaje",
"Mount not authorized by the operation": "Montaje no autorizado por la operación",
"Mount options:": "Opciones de montaje:",
"Mount or create it with its data and run the recovery again.": "Móntalo o créalo con sus datos y vuelve a lanzar la recuperación.",
"Mount path must be an absolute path starting with /": "La ruta de montaje debe ser una ruta absoluta que comience con /",
"Mount path:": "Ruta de montaje:",
"Mount paths must not overlap": "Las rutas de montaje no deben solaparse",
@@ -3750,6 +3766,7 @@
"NVIDIA refresh validated; the container is stopped and its settings are kept": "Actualización de NVIDIA validada; el contenedor está detenido y se conserva su configuración",
"NVIDIA runtime (device and host driver libraries)": "Runtime NVIDIA (dispositivo y bibliotecas del controlador del host)",
"NVIDIA runtime libraries or components are missing": "Faltan bibliotecas o componentes de tiempo de ejecución de NVIDIA",
"NVIDIA runtime rebuilt for the driver of this host": "Runtime de NVIDIA reconstruido para el driver de este host",
"NVIDIA selection not supported by this profile": "Selección NVIDIA no compatible con este perfil",
"NVIDIA services stopped and disabled.": "Los servicios de NVIDIA se detuvieron y deshabilitaron.",
"NVIDIA udev rules and persistence service installed.": "Reglas de NVIDIA udev y servicio de persistencia instalados.",
@@ -4077,6 +4094,7 @@
"No recent": "No reciente",
"No recent Samba servers found.": "No se encontraron servidores Samba recientes.",
"No registered OCI containers are available for selection on this host.": "No hay contenedores OCI registrados disponibles para seleccionar en este host.",
"No restored OCI application is waiting to be recovered.": "No hay ninguna aplicación OCI restaurada pendiente de recuperar.",
"No routing information found.": "No se encontró información de ruta.",
"No scheduled backup jobs configured.": "No se han configurado trabajos de copia de seguridad programados.",
"No scheduled backup jobs found.": "No se encontraron tareas de copia de seguridad programadas.",
@@ -4108,6 +4126,7 @@
"No updates available — run a scan first or wait for the Monitor to refresh.": "No hay actualizaciones disponibles: primero ejecute un análisis o espere a que se actualice el monitor.",
"No updates or failed to fetch templates": "No hay actualizaciones o no se pudieron recuperar las plantillas",
"No usable GPU was found on this host. Immich will be installed on the CPU.": "No se ha encontrado ninguna GPU utilizable en este host. Immich se instalará en la CPU.",
"No usable GPU was found on this host. Recognition stays on the CPU.": "No se ha encontrado ninguna GPU utilizable en este host. El reconocimiento sigue en la CPU.",
"No usable GPU was found on this host. The application will be installed without hardware acceleration.": "No se ha encontrado ninguna GPU utilizable en este host. La aplicación se instalará sin aceleración por hardware.",
"No user groups found.": "No se encontraron grupos de usuarios.",
"No user-created disk storage or fstab mount found.": "No se encontró ningún almacenamiento en disco creado por el usuario ni montaje fstab.",
@@ -4530,6 +4549,7 @@
"Preparing pending restore (network-safe)": "Preparando restauración pendiente (segura para la red)",
"Preparing staging area...": "Preparando el área de preparación...",
"Preparing the NVIDIA GPU...": "Preparando la GPU NVIDIA...",
"Preparing the machine learning container for the new choice...": "Preparando el contenedor Machine learning para la nueva opción...",
"Preparing the recreation...": "Preparando la recreación...",
"Preparing the update...": "Preparando la actualización...",
"Preserving logs to /var/log.hdd before unmounting...": "Preservando registros en /var/log.hdd antes de desmontar...",
@@ -4690,6 +4710,7 @@
"RAM in MiB": "RAM en MiB",
"REPAIR SUMMARY": "RESUMEN DE REPARACIÓN",
"REQUIREMENTS:": "REQUISITOS:",
"ROCm does not support this GPU officially. The application can use it by presenting it as another generation of its family: it is faster than the CPU, but under load the GPU can stop responding, and then the application does not answer until its container is restarted. The ROCm image is also larger than the others.": "ROCm no soporta oficialmente esta GPU. La aplicación puede usarla presentándola como otra generación de su familia: es más rápida que la CPU, pero bajo carga la GPU puede dejar de responder, y entonces la aplicación no contesta hasta que se reinicia su contenedor. Además, la imagen de ROCm es más grande que las demás.",
"ROCm requires /dev/kfd on the host": "ROCm requiere /dev/kfd en el host",
"ROCm requires the render device of an AMD GPU": "ROCm requiere el dispositivo de render de una GPU AMD",
"ROM dump not available — configuring without romfile.": "Volcado de ROM no disponible: configuración sin archivo rom.",
@@ -4757,7 +4778,12 @@
"Recent Samba server": "Servidor Samba reciente",
"Recent logs:": "Registros recientes:",
"Recent test results:": "Resultados de pruebas recientes:",
"Recognition": "Reconocimiento",
"Recognition already runs on that choice": "El reconocimiento ya funciona con esa opción",
"Recognition is not offered on the AMD GPU; video transcoding is.": "El reconocimiento no se ofrece en la GPU AMD; la transcodificación de vídeo sí.",
"Recognition of Immich changed; the application was updated with the new choice.": "Reconocimiento de Immich cambiado; la aplicación se ha actualizado con la nueva opción.",
"Recognition on AMD uses ROCm. Its image is several times larger than the others, so the first installation takes longer, and whether a GPU works with it depends on its model.": "El reconocimiento en AMD usa ROCm. Su imagen es varias veces mayor que las demás, por lo que la primera instalación tarda más, y que una GPU funcione con ella depende de su modelo.",
"Recognition on AMD uses ROCm. Its image is several times larger than the others, so the first installation takes longer.": "El reconocimiento en AMD usa ROCm. Su imagen es varias veces más grande que las demás, así que la primera instalación tarda más.",
"Recognition profile not implemented": "Perfil de reconocimiento no implementado",
"Recognition runs on the CPU.": "El reconocimiento se ejecuta en la CPU.",
"Recommendation: reformat the disk to ext4 for a robust setup — see docs.": "Recomendación: vuelva a formatear el disco a ext4 para una configuración sólida; consulte los documentos.",
@@ -4775,14 +4801,16 @@
"Recover OCI stack": "Recuperar la pila OCI",
"Recover now?": "¿Recuperar ahora?",
"Recover or complete the operation?": "¿Recuperar o completar la operación?",
"Recover restored OCI applications": "Recuperar aplicaciones OCI restauradas",
"Recover the keyfile using your recovery passphrase?": "¿Recuperar el archivo clave usando su frase de contraseña de recuperación?",
"Recover the previous installation": "Recuperar la instalación anterior",
"Recover them now?": "¿Recuperarlos ahora?",
"Recoverable:": "Recuperable:",
"Recovering the previous installation": "Recuperando la instalación anterior",
"Recovery blob upload failed — main backup is OK, but keyfile recovery from PBS will not be available for this backup.": "Error en la carga del blob de recuperación: la copia de seguridad principal está bien, pero la recuperación del archivo clave de PBS no estará disponible para esta copia de seguridad.",
"Recovery blob:": "blob de recuperación:",
"Recovery completed. The container had not been modified yet.": "Recuperación completada. El contenedor aún no había sido modificado.",
"Recovery completed. The displaced disks and the backup are kept; nothing was deleted automatically.": "Recuperación completada. Los discos desplazados y el backup se mantienen; nada fue eliminado automáticamente.",
"Recovery completed. The disks of the failed attempt were removed; the backup is kept.": "Recuperación completada. Los discos del intento fallido se han eliminado; el backup se mantiene.",
"Recovery failed": "La recuperación falló",
"Recovery passphrase": "frase de contraseña de recuperación",
"Recovery setup failed": "Error en la configuración de recuperación",
@@ -5016,6 +5044,7 @@
"Restore plan": "plan de restauración",
"Restore plan summary": "Resumen del plan de restauración",
"Restore source location": "Restaurar ubicación de origen",
"Restored OCI applications": "Aplicaciones OCI restauradas",
"Restored config is on disk; reboot the host to apply.": "La configuración restaurada está en el disco; reinicie el host para aplicar.",
"Restored installation checked": "Instalación restaurada",
"Restored original /bin/gzip": "Restaurado original /bin/gzip",
@@ -5199,6 +5228,7 @@
"Saving the new configuration": "Salvando la nueva configuración",
"Saving the new configuration of the application...": "Guardando la nueva configuración de la aplicación...",
"Saving the new configuration...": "Guardando la nueva configuración...",
"Saving the record of the application...": "Guardando el registro de la aplicación...",
"Saving the stack records...": "Guardando los registros de la pila...",
"Scan storage for new content": "Escanear el almacenamiento en busca de contenido nuevo",
"Scanning available physical disks...": "Escaneando discos físicos disponibles...",
@@ -5662,12 +5692,14 @@
"Start on boot enabled": "Iniciar al arrancar habilitado",
"Start on boot enabled (onboot=1)": "Iniciar al arrancar habilitado (onboot=1)",
"Start on boot will be disabled on source VM only if currently enabled": "El inicio al arrancar se deshabilitará en la máquina virtual de origen solo si está actualmente habilitado",
"Start order of the dependencies restored": "Orden de arranque de las dependencias restaurado",
"Start scrub for a ZFS pool": "Iniciar limpieza para un grupo ZFS",
"Start short self-test (~2 min)": "Iniciar una autoprueba breve (~2 min)",
"Start terminal multiplexer (recommended):": "Iniciar multiplexor de terminal (recomendado):",
"Start the LXC when finished to apply the selected configuration?": "¿Iniciar el LXC al terminar para aplicar la configuración seleccionada?",
"Start the VM": "Inicie la máquina virtual",
"Start the VM to begin Windows installation from the mounted ISO.": "Inicie la VM para comenzar la instalación de Windows desde la ISO montada.",
"Start the applications once they are registered? Answer No if the original containers are still running on another host: both would use the same addresses.": "¿Iniciar las aplicaciones una vez registradas? Responde No si los contenedores originales siguen en marcha en otro host: los dos usarían las mismas direcciones.",
"Start the converted container:": "Inicie el contenedor convertido:",
"Start the main system upgrade:": "Inicie la actualización principal del sistema:",
"Start the stack with Proxmox": "Iniciar la pila con Proxmox",
@@ -5884,6 +5916,7 @@
"That VM is currently stopped, so the GPU can be reassigned now.": "Esa VM está actualmente detenida, por lo que la GPU se puede reasignar ahora.",
"That doesn't look like an SSH private key. Pick the private key file (no .pub extension, parseable by ssh-keygen).": "Eso no parece una clave privada SSH.Elige el archivo de clave privada (sin extensión .pub, analizable mediante ssh-keygen).",
"The .conf files under /config/fail2ban are rewritten on every start. Keep customizations in the matching .local file, for example jail.local for jail.conf.": "Los archivos .conf bajo /config/fail2ban son reescritos en cada inicio. Mantenga las personalizaciones en el archivo .local coincidente, por ejemplo la cárcel.local para jail.conf.",
"The AMD GPU did not complete a test inference with ROCm": "La GPU AMD no completó una inferencia de prueba con ROCm",
"The AMD driver does not offer its compute interface (/dev/kfd) on this host.": "El driver de AMD no ofrece su interfaz de cómputo (/dev/kfd) en este host.",
"The AppArmor/seccomp relaxation does not include the required consent": "La relajación AppArmor/seccomp no incluye el consentimiento necesario",
"The Bookmark Everything App": "La aplicación Todo Marcador",
@@ -5940,6 +5973,8 @@
"The PostgreSQL volume needs at least 8 GB": "El volumen de PostgreSQL necesita al menos 8 GB",
"The Proxmox archive keyring is missing; Ceph installation cannot continue safely": "Falta el conjunto de claves de archivo Proxmox;La instalación de Ceph no puede continuar de forma segura",
"The Proxmox inventory and the local configurations differ": "El inventario de Proxmox y las configuraciones locales difieren",
"The ROCm image has no support for the AMD GPU of this host:": "La imagen de ROCm no tiene soporte para la GPU AMD de este host:",
"The ROCm image of this application has no support for the AMD GPU of this host:": "La imagen de ROCm de esta aplicación no tiene soporte para la GPU AMD de este host:",
"The Rclone configuration needs at least 1 GB": "La configuración Rclone necesita al menos 1 GB",
"The Rclone rootfs needs at least 2 GB": "El rootfs de Rclone necesita al menos 2 GB",
"The Selkies profile requires a verified LinuxServer image": "El perfil de Selkies requiere una imagen verificada LinuxServer",
@@ -5974,12 +6009,14 @@
"The administrator user contains characters that are not allowed": "El usuario administrador contiene caracteres que no se permiten",
"The all-in-one AI application.": "La aplicación de IA todo en uno.",
"The application configuration needs a first start to complete.": "La configuración de la aplicación necesita un primer arranque para completarse.",
"The application could not be recovered:": "No se pudo recuperar la aplicación:",
"The application could not be recreated:": "No se pudo recrear la aplicación:",
"The application did not complete its initial setup:": "La aplicación no completó su configuración inicial:",
"The application did not get an address on the access network:": "La aplicación no obtuvo una dirección en la red de acceso:",
"The application did not pass its HTTP check:": "La aplicación no superó la comprobación HTTP:",
"The application did not respond in time:": "La aplicación no respondió a tiempo:",
"The application has been recreated with the new options.": "La aplicación se ha recreado con las nuevas opciones.",
"The application needs it for": "La aplicación lo necesita para",
"The application stopped during its first start:": "La aplicación se detuvo durante su primer comienzo:",
"The application was removed": "La aplicación se ha eliminado",
"The archive could not be extracted.": "No se pudo extraer el archivo.",
@@ -6169,7 +6206,10 @@
"The kernel module is not active:": "El módulo del núcleo no está activo:",
"The kernel module of version": "El módulo del kernel de la versión.",
"The local envelope is dropped and future backups do not upload anything. Uploaded envelopes already on PBS stay intact and remain recoverable with their original passphrase.": "El sobre local se elimina y las copias de seguridad futuras no cargan nada. Los sobres cargados que ya están en PBS permanecen intactos y recuperables con su frase de contraseña original.",
"The local storage does not accept snippets": "El almacenamiento local no admite snippets",
"The long test runs directly on the disk hardware.": "La prueba larga se ejecuta directamente en el hardware del disco.",
"The machine learning container is rebuilt with the image of the new choice. Its model cache, the library and the database are kept. The whole application is stopped and updated, as in an update; if anything fails, the previous containers and the previous choice are restored.": "El contenedor Machine learning se reconstruye con la imagen de la nueva opción. Se conservan su caché de modelos, la biblioteca y la base de datos. Toda la aplicación se detiene y se actualiza, como en una actualización; si algo falla, se restauran los contenedores y la opción anteriores.",
"The main container of this application is not on this host. Restore it too.": "El contenedor principal de esta aplicación no está en este host. Restáuralo también.",
"The main member must stop first and start last": "El miembro principal debe parar primero y comenzar el último",
"The main member of the stack is missing": "Falta el miembro principal de la pila",
"The managed host firewall rule could not be removed and was left unchanged.": "No se pudo eliminar la regla del cortafuegos del host gestionada por ProxMenux; se ha dejado sin cambios.",
@@ -6231,6 +6271,7 @@
"The prepared directory escapes the rootfs:": "El directorio preparado escapa a los rootfs:",
"The preselected VMID does not exist on this host:": "El VMID preseleccionado no existe en este host:",
"The previous native backup will be restored. Shared host directories are not reverted. Displaced disks are kept.": "Se restaurará el backup nativo anterior. Los directorios compartidos del host no se revierten. Los discos desplazados se conservan.",
"The previous recognition choice could not be put back:": "No se pudo volver a la opción de reconocimiento anterior:",
"The previous stack contract is not safe; review it before reusing it": "El contrato de pila anterior no es seguro; revíselo antes de reutilizarlo",
"The previous stack contract is not valid; it is not archived automatically": "El contrato de pila anterior no es válido; no se archiva automáticamente",
"The previous stack contract still has containers or VMs:": "El contrato de pila anterior todavía tiene contenedores o VMs:",
@@ -6240,6 +6281,7 @@
"The private network allocator was not found": "El aleator de la red privada no fue encontrado",
"The private network is still used by another container and is kept:": "Otro contenedor sigue usando la red privada, así que se conserva:",
"The private network must be assigned automatically": "La red privada debe ser asignada automáticamente",
"The private network of the application is already in use on this host:": "La red privada de la aplicación ya está en uso en este host:",
"The privileged deployment does not include the required explicit consent": "El despliegue privilegiado no incluye el consentimiento explícito necesario",
"The prlimit soft value exceeds the hard value": "El valor prlimit blando supera el valor duro",
"The proposal changes the identity of the instance": "La propuesta cambia la identidad de la instancia",
@@ -6250,6 +6292,7 @@
"The read-only view was not published": "La opinión de sólo lectura no fue publicada",
"The read/write view was not published": "La vista de lectura/escritura no se publicó",
"The recipe requires configuration at startup; its coordinated replay is not available": "La configuración de la receta requiere al inicio; su reproducción coordinada no está disponible",
"The recognition of Immich was not changed:": "No se cambió el reconocimiento de Immich:",
"The record belongs to another container": "El registro pertenece a otro contenedor",
"The record does not belong to this operation": "El registro no pertenece a esta operación",
"The record no longer belongs to this operation": "El registro ya no pertenece a esta operación",
@@ -6262,6 +6305,8 @@
"The remote must already be created and authorized in the Rclone web UI. This operation restarts the CT and publishes two FUSE views on the host.": "El remoto ya debe estar creado y autorizado en la interfaz web de Rclone. Esta operación reinicia el CT y publica dos puntos de vista FUSE sobre el host.",
"The remote path must be relative and cannot contain line breaks": "La ruta remota debe ser relativa y no puede contener saltos de línea",
"The removal could not be prepared:": "No se pudo preparar la eliminación:",
"The render device does not belong to the GPU of that choice": "El dispositivo de render no pertenece a la GPU de esa opción",
"The render device does not exist:": "El dispositivo de render no existe:",
"The repair must preserve the image dependencies:": "La reparación debe preservar las dependencias de la imagen:",
"The requested VMID block is already in use": "El bloque VMID solicitado ya está en uso",
"The requested machine learning GPU profile is not working; it is not replaced by CPU": "El perfil de GPU solicitado para Machine learning no funciona; no se sustituye por CPU",
@@ -6395,6 +6440,7 @@
"These VMIDs are not free:": "Estos VMID no están libres:",
"These are the changes that will be made": "Estos son los cambios que se harán",
"These container mount paths have an extension-like suffix:": "Estas rutas de montaje del contenedor tienen un sufijo parecido a una extensión:",
"These containers were restored from a backup and this host has no record of their OCI application:": "Estos contenedores se restauraron desde un backup y este host no tiene el registro de su aplicación OCI:",
"These interface configurations will be removed": "Estas configuraciones de interfaz serán eliminadas.",
"These manual changes cannot be adopted safely:": "Estos cambios manuales no se pueden incorporar con seguridad:",
"These paths will not be restored live and will be extracted for manual recovery.": "Estas rutas no se restaurarán en vivo y se extraerán para su recuperación manual.",
@@ -6405,6 +6451,8 @@
"This NFS share is fully restricted — even the host root cannot write to it.": "Este recurso compartido NFS está completamente restringido: ni siquiera la raíz del host puede escribir en él.",
"This VM requires extra installation steps, see install guide at:\nhttps://github.com/community-scripts/ProxmoxVE/discussions/144": "Esta máquina virtual requiere pasos de instalación adicionales; consulte la guía de instalación en:\nhttps://github.com/community-scripts/ProxmoxVE/discussions/144",
"This action will:": "Esta acción:",
"This application cannot be recovered yet; nothing was changed:": "Esta aplicación todavía no se puede recuperar; no se ha cambiado nada:",
"This application is not an Immich installed by ProxMenux": "Esta aplicación no es un Immich instalado por ProxMenux",
"This archive does not contain a recognized backup layout.": "Este archivo no contiene un diseño de copia de seguridad reconocido.",
"This backup does not contain any restorable paths.": "esta copia de seguridad no contiene rutas restaurables.",
"This backup includes /etc/zfs/zpool.cache (host-specific ZFS state).": "esta copia de seguridad incluye /etc/zfs/zpool.cache (estado ZFS específico del host).",
@@ -6418,6 +6466,7 @@
"This container has no GPU configured. Coral TPU works best alongside hardware video decoding (Quick Sync, VA-API, NVENC) for apps like Frigate.": "Este contenedor no tiene GPU configurada. Coral TPU funciona mejor junto con la decodificación de video por hardware (Quick Sync, VA-API, NVENC) para aplicaciones como Frigate.",
"This container is not a member of a multi-container application": "Este contenedor no forma parte de una aplicación de varios contenedores",
"This container is not a registered OCI instance.": "Este contenedor no es una instancia OCI registrada.",
"This container was restored or came back from another host after its record on this host was written. Open this menu again to recover it; nothing was changed.": "Este contenedor se restauró o volvió de otro host después de que se escribiera su registro en este host. Abre de nuevo este menú para recuperarlo; no se ha cambiado nada.",
"This converts all directory UIDs/GIDs by adding 100000": "Esto convierte todos los UID/GID del directorio agregando 100000",
"This converts all file UIDs/GIDs by adding 100000": "Esto convierte todos los archivos UID/GID agregando 100000",
"This creates a backup in case you need to revert changes": "Esto crea una copia de seguridad en caso de que necesite revertir los cambios.",
@@ -6721,6 +6770,7 @@
"Unsupported storage mode:": "Modo de almacenamiento sin soporte:",
"Unsupported tmpfs options": "Opciones de tmpfs sin soporte",
"Unsupported volume options:": "Opciones de volumen no admitidas:",
"Until they are registered again they cannot be updated or managed from here. The recovery checks first that everything they need is on this host and changes nothing otherwise.": "Hasta que se registren de nuevo no se pueden actualizar ni gestionar desde aquí. La recuperación comprueba antes que este host tiene todo lo que necesitan y, si no es así, no cambia nada.",
"Untrusted or modified NVIDIA hook": "Gancho NVIDIA sin confianza o modificado",
"Unused image removed from the cache:": "Imagen sin uso eliminada de la caché:",
"Unused images removed from the cache:": "Imágenes sin uso eliminadas de la caché:",
@@ -6824,6 +6874,7 @@
"Use option 1 to add a local disk (as Proxmox storage and/or as a host mount).": "Utilice la opción 1 para agregar un disco local (como almacenamiento Proxmox y/o como montaje de host).",
"Use option 1 to add an iSCSI target as Proxmox storage.": "Utilice la opción 1 para agregar un destino iSCSI como almacenamiento Proxmox.",
"Use pigz for faster gzip compression": "Usar pigz para comprimir gzip más rápido",
"Use the GPU with ROCm anyway?": "¿Usar la GPU con ROCm de todos modos?",
"Use the Guided Cleanup option to fix issues safely": "Utilice la opción Limpieza guiada para solucionar problemas de forma segura",
"Use the Guided Repair option to fix issues safely": "Utilice la opción Reparación guiada para solucionar problemas de forma segura",
"Use these commands on your Proxmox host to access an LXC container's terminal:": "Utilice estos comandos en su host Proxmox para acceder a la terminal de un contenedor LXC:",
@@ -7056,6 +7107,8 @@
"Weixin (WeChat) is an instant messaging, social media, and mobile payment app developed by Tencent.": "Weixin (WeChat) es una aplicación de mensajería instantánea, redes sociales y pago móvil desarrollada por Tencent.",
"What cannot be translated:": "Lo que no puede traducirse:",
"What do you want to do?": "¿Qué es lo que quieres hacer?",
"What runs the recognition of Immich": "Qué ejecuta el reconocimiento de Immich",
"What to recreate": "Qué recrear",
"What would you like to do?": "¿Qué te gustaría hacer?",
"When asked to select a disk, click Load Driver and load the VirtIO drivers.": "Cuando se le solicite seleccionar un disco, haga clic en Cargar controlador y cargue los controladores VirtIO.",
"When this GPU is passed through to a VM, the Proxmox host will lose all video output on the physical monitor.": "Cuando esta GPU pasa a una VM, el host Proxmox perderá toda la salida de video en el monitor físico.",
@@ -7180,6 +7233,7 @@
"Znc is an IRC network bouncer or BNC. It can detach the client from the actual IRC server, and also from selected channels. Multiple clients from different locations can connect to a single ZNC account simultaneously and therefore appear under the same nickname on IRC.": "Znc es un rebote de red IRC o BNC. Puede separar al cliente del servidor IRC real, y también de canales seleccionados. Múltiples clientes de diferentes ubicaciones pueden conectarse a una sola cuenta ZNC simultáneamente y por lo tanto aparecen bajo el mismo apodo en IRC.",
"Zotero is a free, easy-to-use tool to help you collect, organize, annotate, cite, and share research.": "Zotero es una herramienta gratuita, fácil de usar para ayudarle a recoger, organizar, anotar, citar y compartir la investigación.",
"a device it asks for cannot be translated:": "un dispositivo que solicita no se puede traducir:",
"a different host monitor include already exists:": "ya existe un include de monitor del host distinto:",
"a value is required": "se necesita un valor",
"aMule WebUI (password only, no username)": "aMule WebUI (sólo contraseña, sin nombre de usuario)",
"aMule is a multi-platform client for the ED2K file sharing network and based on the windows client eMule. aMule started in August 2003, as a fork of xMule, which is a fork of lMule.": "aMule es un cliente multiplataforma para la red de intercambio de archivos ED2K y basado en el eMule cliente de ventanas. aMule comenzó en agosto de 2003, como un fork de xMule, que es un fork de lMule.",
@@ -7199,6 +7253,8 @@
"and": "y",
"and could not be unmounted — disk may be busy.": "y no se pudo desmontar; es posible que el disco esté ocupado.",
"and upgrade the system (disables the enterprise repo)": "y actualizar el sistema (deshabilita el repositorio empresarial)",
"and, read-only, at": "y, en solo lectura, en",
"another container of this host already uses the address": "otro contenedor de este host ya usa la dirección",
"apex device node not found yet; a reboot may be required.": "el nodo del dispositivo Apex aún no se ha encontrado; Es posible que sea necesario reiniciar.",
"apex group and udev rules are in place.": "Las reglas del grupo Apex y de udev están vigentes.",
"apex group removed.": "grupo ápice eliminado.",
@@ -7270,6 +7326,7 @@
"empty = generate": "vacío = generar",
"exFAT (portable: Windows/Linux/macOS)": "exFAT (portátil: Windows/Linux/macOS)",
"exFAT tools installed successfully.": "Herramientas exFAT instaladas correctamente.",
"experimental on this GPU": "experimental en esta GPU",
"ext4 — Proxmox dir storage (recommended)": "ext4 — Almacenamiento de directorios de Proxmox (recomendado)",
"ext4 — recommended, most compatible": "ext4: recomendado, más compatible",
"fail2ban-client could not communicate with the server": "El cliente fail2ban no pudo comunicarse con el servidor.",
@@ -7381,11 +7438,24 @@
"is not configured as machine type q35.": "no está configurado como tipo de máquina q35.",
"is not in the patch.sh supported list. The patch may no-op or fail; review keylase/nvidia-patch README before continuing.": "no está en la lista compatible con patch.sh. Es posible que el parche no funcione o falle; revise el archivo README de keylase/nvidia-patch antes de continuar.",
"is not supported by the official Google libedgetpu APT repository.": "no es compatible con el repositorio oficial de Google libedgetpu APT.",
"is now": "ahora es",
"is one of the": "es uno de los",
"is referenced in the following stopped VM(s)/CT(s):": "se hace referencia en las siguientes VM/CT detenidas:",
"it asks for the network of the host; the container gets its own address instead": "pide la red del host; el contenedor obtiene su propia dirección en su lugar",
"it could not be started:": "no se pudo iniciar:",
"it is locked or its disk could not be mounted": "está bloqueado o no se pudo montar su disco",
"it publishes no other architecture": "no publica ninguna otra arquitectura",
"it uses an NVIDIA GPU and this host has no working NVIDIA driver and Container Toolkit. Install them and run the recovery again.": "usa una GPU NVIDIA y este host no tiene el driver de NVIDIA y el Container Toolkit en funcionamiento. Instálalos y vuelve a lanzar la recuperación.",
"it uses the Compose option": "utiliza la opción Compose",
"its NVIDIA runtime could not be rebuilt for the driver of this host; the container will not start until it is installed again:": "no se pudo reconstruir su runtime de NVIDIA para el driver de este host; el contenedor no arrancará hasta que se instale de nuevo:",
"its Rclone mount is published again on this host at": "su montaje de Rclone se publica de nuevo en este host en",
"its Rclone mount is published by a host script that a backup does not include. Enable the mount again from the Rclone entry of the catalog before starting it.": "su montaje de Rclone lo publica un script del host que un backup no incluye. Activa de nuevo el montaje desde la entrada de Rclone del catálogo antes de iniciarlo.",
"its backup was made before ProxMenux kept the record inside the container": "su backup es anterior a que ProxMenux guardara el registro dentro del contenedor",
"its configuration could not be read": "no se pudo leer su configuración",
"its hookscript does not exist on this host:": "su hookscript no existe en este host:",
"its host firewall rule was not added:": "no se añadió su regla del firewall del host:",
"its include exists with other content:": "su include existe con otro contenido:",
"its record is missing and it was not found among the restored containers": "falta su registro y no está entre los contenedores restaurados",
"journald MaxLevelStore is adequate for auth logging": "journald MaxLevelStore es adecuado para el registro de autenticación",
"journald drop-in created: /etc/systemd/journald.conf.d/proxmenux-loglevel.conf": "drop-in de diario creado: /etc/systemd/journald.conf.d/proxmenux-loglevel.conf",
"journald log level restored": "Nivel de registro de journald restaurado",
@@ -7552,6 +7622,7 @@
"server?": "¿servidor?",
"servers found on the network.": "servidores encontrados en la red.",
"servers found.": "servidores encontrados.",
"several containers of this host carry the same installation:": "varios contenedores de este host llevan la misma instalación:",
"sha256sum not found. Cannot verify Borg binary.": "sha256sum no encontrado. No se puede verificar el binario Borg.",
"shadPS4 is an early PlayStation 4 emulator for Windows, Linux and macOS written in C++.": "shadPS4 es un emulador de PlayStation 4 temprano para Windows, Linux y macOS escrito en C++.",
"showmount command is not working properly.": "El comando showmount no funciona correctamente.",
@@ -7583,10 +7654,24 @@
"systemctl restart networking failed:": "systemctl restart networking falló:",
"systemd OnCalendar expression": "expresión systemd OnCalendar",
"the API key was not generated on the first start": "la clave de API no se generó en el primer comienzo",
"the NVIDIA hook of the container is not available:": "el hook de NVIDIA del contenedor no está disponible:",
"the bridge does not exist on this host:": "el bridge no existe en este host:",
"the bridge exists on this host with another network:": "el bridge existe en este host con otra red:",
"the container has its own address, so the port Docker published on the host is not needed": "el contenedor tiene su propia dirección, por lo que el puerto Docker publicado en el host no es necesario",
"the containers do not agree on the private network of": "los contenedores no coinciden en la red privada de",
"the device does not exist on this host:": "el dispositivo no existe en este host:",
"the host directory does not exist:": "el directorio del host no existe:",
"the host firewall rule of the installation is not added again; allow its port in the firewall of this host if you use it.": "la regla del firewall del host de la instalación no se vuelve a añadir; permite su puerto en el firewall de este host si lo usas.",
"the qBittorrent schema is not available": "el esquema qBittorrent no está disponible",
"the record it carries does not belong to this container": "el registro que lleva no pertenece a este contenedor",
"the record of the application does not list this container": "el registro de la aplicación no incluye este contenedor",
"the settings of its include are not in the record:": "los ajustes de su include no están en el registro:",
"the start order of its dependencies is not in the record": "el orden de arranque de sus dependencias no está en el registro",
"this configuration needs the device": "esta configuración necesita el dispositivo",
"this container of the application is not on this host. Restore it too.": "este contenedor de la aplicación no está en este host. Restáuralo también.",
"this distribution": "esta distribución",
"this host keeps a record of another installation with an operation pending": "este host guarda un registro de otra instalación con una operación pendiente",
"this host keeps an unreadable record for this ID": "este host guarda un registro ilegible para este ID",
"tmpfs size in MB for": "tamaño de tmpfs en MB para",
"tmpfs size too small for": "tamaño de tmpfs demasiado pequeño para",
"to": "a",
@@ -7597,6 +7682,8 @@
"to sharedfiles group": "al grupo de archivos compartidos",
"total": "total",
"umount the path if currently mounted": "desmontar la ruta si actualmente está montada",
"unknown include:": "include desconocido:",
"unknown mount hook:": "hook de montaje desconocido:",
"unprivileged LXC": "LXC sin privilegios",
"unsupported credential generator": "generador de credenciales no admitido",
"unsupported dependency condition": "condición de dependencia no admitida",
+2 -2
View File
@@ -116,7 +116,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 32
}
},
@@ -132,7 +132,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 32
}
}
+2 -1
View File
@@ -117,7 +117,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 100
}
}
@@ -887,6 +887,7 @@
{
"id": "rocm",
"label": "AMD (VA-API and ROCm detection)",
"amd_gfx_targets": [100300, 110000, 110001, 110002, 110500, 110501, 120000, 120001],
"architectures": [
"amd64"
],
+2 -2
View File
@@ -87,7 +87,7 @@
],
"default": "managed-volume",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 8
}
}
@@ -190,7 +190,7 @@
"container_path": "/cache",
"mode": "managed-volume",
"user_selectable": false,
"backup": false,
"backup": true,
"shared_with_other_lxc": false,
"source_path": null,
"source_path_prompt": null
+5 -5
View File
@@ -366,7 +366,7 @@
"container_path": "/cache",
"mode": "managed-volume",
"user_selectable": false,
"backup": false,
"backup": true,
"shared_with_other_lxc": false,
"source_path": null,
"source_path_prompt": null
@@ -638,8 +638,8 @@
{
"pci_id": "1002:164c",
"gpu": "AMD Lucienne integrated graphics",
"reason": "Real buffalo_l inference caused SDMA/compute timeouts and repeated host GPU resets.",
"tested_on": "2026-08-26"
"reason": "The ROCm image carries no kernels for gfx90c: every inference aborts in rocBLAS. The installer reads the generation from the compute driver and keeps recognition on the CPU.",
"tested_on": "2026-10-03"
}
]
},
@@ -1208,8 +1208,8 @@
{
"pci_id": "1002:164c",
"gpu": "AMD Lucienne integrated graphics",
"reason": "Real buffalo_l inference caused SDMA/compute timeouts and repeated host GPU resets.",
"tested_on": "2026-08-26"
"reason": "The ROCm image carries no kernels for gfx90c: every inference aborts in rocBLAS. The installer reads the generation from the compute driver and keeps recognition on the CPU.",
"tested_on": "2026-10-03"
}
],
"warning": "Do not enable automatically on unsupported AMD integrated GPUs. HSA overrides are manual compatibility experiments, not a validated default."
+2 -2
View File
@@ -100,7 +100,7 @@
],
"default": "managed-volume",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 8
}
},
@@ -116,7 +116,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 32
}
}
+2 -2
View File
@@ -132,7 +132,7 @@
],
"default": "managed-volume",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 16
}
},
@@ -148,7 +148,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 32
}
}
+2 -2
View File
@@ -116,7 +116,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 32
}
},
@@ -132,7 +132,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 32
}
}
+2 -1
View File
@@ -117,7 +117,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 100
}
}
@@ -887,6 +887,7 @@
{
"id": "rocm",
"label": "AMD (VA-API and ROCm detection)",
"amd_gfx_targets": [100300, 110000, 110001, 110002, 110500, 110501, 120000, 120001],
"architectures": [
"amd64"
],
+5 -5
View File
@@ -366,7 +366,7 @@
"container_path": "/cache",
"mode": "managed-volume",
"user_selectable": false,
"backup": false,
"backup": true,
"shared_with_other_lxc": false,
"source_path": null,
"source_path_prompt": null
@@ -638,8 +638,8 @@
{
"pci_id": "1002:164c",
"gpu": "AMD Lucienne integrated graphics",
"reason": "Real buffalo_l inference caused SDMA/compute timeouts and repeated host GPU resets.",
"tested_on": "2026-08-26"
"reason": "The ROCm image carries no kernels for gfx90c: every inference aborts in rocBLAS. The installer reads the generation from the compute driver and keeps recognition on the CPU.",
"tested_on": "2026-10-03"
}
]
},
@@ -1208,8 +1208,8 @@
{
"pci_id": "1002:164c",
"gpu": "AMD Lucienne integrated graphics",
"reason": "Real buffalo_l inference caused SDMA/compute timeouts and repeated host GPU resets.",
"tested_on": "2026-08-26"
"reason": "The ROCm image carries no kernels for gfx90c: every inference aborts in rocBLAS. The installer reads the generation from the compute driver and keeps recognition on the CPU.",
"tested_on": "2026-10-03"
}
],
"warning": "Do not enable automatically on unsupported AMD integrated GPUs. HSA overrides are manual compatibility experiments, not a validated default."
+2 -2
View File
@@ -100,7 +100,7 @@
],
"default": "managed-volume",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 8
}
},
@@ -116,7 +116,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 32
}
}
+2 -2
View File
@@ -132,7 +132,7 @@
],
"default": "managed-volume",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 16
}
},
@@ -148,7 +148,7 @@
],
"default": "host-bind",
"managed_volume": {
"backup": false,
"backup": true,
"default_size_gb": 32
}
}
+6
View File
@@ -176,6 +176,12 @@ oci_quiet pct unmount "$VMID"
sed -i -E '/^lxc\.environment\.runtime: RCLONE_RC_(USER|PASS)=/d' "/etc/pve/lxc/${VMID}.conf"
oci_quiet pct set "$VMID" --entrypoint /usr/local/bin/rclone-mount-lxc-start
sed -i -E '/^hookscript:/d' "/etc/pve/lxc/${VMID}.conf"
# A new Proxmox installation accepts no snippets on `local`.
if ! pvesm status --content snippets 2>/dev/null | awk 'NR > 1 && $1 == "local" {found=1} END {exit !found}'; then
LOCAL_CONTENT=$(pvesh get /storage/local --output-format json 2>/dev/null | jq -r '.content // empty')
[[ -n $LOCAL_CONTENT ]] || die "$(translate "The local storage does not accept snippets")"
oci_quiet pvesm set local --content "${LOCAL_CONTENT},snippets"
fi
oci_quiet pct set "$VMID" --hookscript "local:snippets/$(basename "$HOOK_PATH")"
msg_ok "$(translate "Mount mode applied")"
+7
View File
@@ -98,6 +98,9 @@ done
[[ -r $STACK_DEPENDENCY_HOOK ]] || die "$(translate "The stack startup hook was not found")"
ML_ACCELERATION=$(jq -er '.machine_learning.acceleration // "cpu"' "$DEPLOYMENT_FILE")
ML_GFX_OVERRIDE=$(jq -r '.machine_learning.gfx_override // empty' "$DEPLOYMENT_FILE")
[[ -z $ML_GFX_OVERRIDE || ( $ML_ACCELERATION == rocm && $ML_GFX_OVERRIDE =~ ^[0-9]{1,2}\.[0-9]\.[0-9]$ ) ]] \
|| die "$(translate "Invalid ROCm generation override")"
source "$SCRIPT_DIR/oci_nvidia_setup.sh"
source "$SCRIPT_DIR/oci_immich_ml.sh"
validate_immich_ml_profile
@@ -437,6 +440,10 @@ set_runtime_env "$ML_ID" MACHINE_LEARNING_CACHE_FOLDER /cache
set_runtime_env "$ML_ID" TRANSFORMERS_CACHE /cache
set_runtime_env "$ML_ID" MACHINE_LEARNING_MODEL_INTRA_OP_THREADS 2
set_runtime_env "$ML_ID" MACHINE_LEARNING_MODEL_INTER_OP_THREADS 1
if [[ -n $ML_GFX_OVERRIDE ]]; then
set_runtime_env "$ML_ID" HSA_OVERRIDE_GFX_VERSION "$ML_GFX_OVERRIDE"
set_runtime_env "$ML_ID" HSA_USE_SVM 0
fi
configure_immich_ml_gpu
oci_quiet pct mount "$ML_ID"
ML_ROOT="/var/lib/lxc/${ML_ID}/rootfs"
+3 -2
View File
@@ -285,7 +285,8 @@ apply_host_monitor_firewall() {
apply_host_monitor() {
[[ -n ${HOST_MONITOR:-} ]] || return 0
# PVE permits lxc.include but not namespace keys directly in the CT config.
# This static, cluster-persistent companion must accompany cross-host restores.
# This static, cluster-persistent companion is written again by the recovery
# of a container restored on another host.
# /etc/pve/proxmenux is the same on every node of a cluster; /etc/pve/lxc
# is the folder of this node only.
local include=/etc/pve/proxmenux/host-monitor native
@@ -303,7 +304,7 @@ apply_host_monitor() {
# Do not remove the Proxmox pre-start, autodev or post-stop hooks.
set_lxc_directive lxc.hook.mount ""
msg_ok "$(translate "Host monitor configured: shared PID and network namespaces, LXCFS disabled in this container")"
msg_info2 "$(translate "Every node of this cluster already has this file. If you restore this container on any other Proxmox host, copy it to the same path first, because the container backup does not include it:") $include"
msg_info2 "$(translate "Every node of this cluster already has this file. A backup of the container does not include it: on another Proxmox host it is written again when the application is recovered from Manage installed OCI applications:") $include"
}
verify_host_monitor() {
+385
View File
@@ -0,0 +1,385 @@
#!/usr/bin/env python3
"""Copies of the record of an OCI application that outlive the host record.
The record of an installation lives on the disk of the node that installed
it. Two copies are kept with it, written together after every operation:
- inside the container, in its root filesystem, so a backup carries it: a
container restored on another Proxmox host, or on this one after a
reinstall, still has what is needed to register it again. The container
could change this copy, so it is checked against the configuration Proxmox
restored before anything is taken from it;
- in /etc/pve, which every node of a cluster shares and only root of the host
reads, so a container that migrates finds its record on the node it moves
to. This one is trusted.
Neither is read back as a source of truth while the host record is current.
"""
from __future__ import annotations
import argparse
import contextlib
import datetime
import json
import os
from pathlib import Path
import re
import socket
import stat
import subprocess
import sys
import uuid
import oci_instances as instances
from oci_installation_state import sha
from oci_ui import translate
DIRECTORY = '.proxmenux'
NAME = 'oci-record.json'
KIND = 'proxmenux.oci-carried-record'
STAMP = 'carried.sha256'
LIMIT = 16 * 1024 * 1024
STACK_CONTRACT = '/etc/pve/priv/proxmenux-stack-{}.json'
NVIDIA_HOOK = re.compile(r'/usr/local/lib/proxmenux/oci/nvidia-mount-([a-f0-9]{64})\.sh')
SNIPPETS = Path('/var/lib/vz/snippets')
# Every node of a cluster reads this folder and only root of the host can:
# a container that moves to another node finds its record there.
CLUSTER = Path('/etc/pve/priv/proxmenux/oci')
RCLONE_HOOK = re.compile(r'^hookscript: local:snippets/(proxmenux-rclone-[0-9]+-fuse-hook\.sh)$', re.MULTILINE)
def _run(*args):
return subprocess.run(args, capture_output=True, text=True, timeout=120, check=False)
def running_pid(vmid):
result = _run('lxc-info', '-n', str(int(vmid)), '-pH')
pid = result.stdout.strip()
return int(pid) if result.returncode == 0 and pid.isdigit() else None
@contextlib.contextmanager
def container_root(vmid, mount_stopped=True):
"""The root filesystem of the container as the host sees it, or None. A
running container is reached through its init process; a stopped one is
mounted for as long as the block lasts."""
pid = running_pid(vmid)
if pid:
yield Path(f'/proc/{pid}/root')
return
if not mount_stopped or _run('pct', 'mount', str(int(vmid))).returncode != 0:
yield None
return
try:
yield Path(f'/var/lib/lxc/{int(vmid)}/rootfs')
finally:
_run('pct', 'unmount', str(int(vmid)))
def mapped_root(config):
"""The host owner that is root inside the container."""
owner = {}
for line in config.splitlines():
fields = line.partition(': ')[2].split()
if line.startswith('lxc.idmap: ') and len(fields) == 4 and fields[1] == '0' and fields[0] in 'ug':
owner.setdefault(fields[0], int(fields[2]))
default = 100000 if re.search(r'^unprivileged: 1$', config, re.MULTILINE) else 0
return owner.get('u', default), owner.get('g', default)
def _directory(root, owner=None):
"""The private folder of the copy, opened without following a link the
container could have left in its place."""
root_fd = os.open(root, os.O_RDONLY | os.O_DIRECTORY)
try:
if owner is not None:
with contextlib.suppress(FileExistsError):
os.mkdir(DIRECTORY, 0o700, dir_fd=root_fd)
fd = os.open(DIRECTORY, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=root_fd)
finally:
os.close(root_fd)
if owner is not None:
os.fchown(fd, *owner)
os.fchmod(fd, 0o700)
return fd
def valid_copy(value):
return (isinstance(value, dict) and value.get('kind') == KIND and value.get('schema_version') == 1
and isinstance(value.get('record'), dict))
def read_cluster(vmid):
"""The copy the cluster keeps for a container, or None."""
path = CLUSTER / f'{int(vmid)}.json'
try:
if path.is_symlink() or path.stat().st_size > LIMIT:
return None
value = json.loads(path.read_text())
except (OSError, ValueError):
return None
return value if valid_copy(value) else None
def write_cluster(vmid, value):
"""Best effort: /etc/pve is read-only without quorum and takes files of
up to 1 MiB. The copy inside the container does not depend on it."""
try:
CLUSTER.mkdir(parents=True, exist_ok=True)
(CLUSTER / f'{int(vmid)}.json').write_text(json.dumps(value))
except OSError:
return False
return True
def remove_cluster(vmid):
with contextlib.suppress(OSError):
(CLUSTER / f'{int(vmid)}.json').unlink()
def keep_cluster(vmid, value, generation):
"""Leave in the cluster the same copy the container carries."""
shared = read_cluster(vmid)
if shared is not None and digest(shared) == digest(value) and shared.get('generation') == generation:
return
write_cluster(vmid, dict(value, generation=generation, saved_at=value.get('saved_at')
or datetime.datetime.now(datetime.timezone.utc).isoformat()))
def replaced_elsewhere(record, record_path, shared, node):
"""Whether another node of the cluster wrote a different record for this
container after the one of this host: the container was changed there and
came back."""
if shared is None or shared.get('node') == node or shared['record'] == record \
or shared['record'].get('installation_id') != record['installation_id'] \
or shared['record'].get('vmid') != record['vmid']:
return False
try:
saved = datetime.datetime.fromisoformat(str(shared.get('saved_at')))
return saved.timestamp() > record_path.stat().st_mtime
except (ValueError, OSError, TypeError):
return False
def read_copy(root):
"""The copy a container carries, or None when it has none that can be used."""
try:
directory = _directory(root)
except OSError:
return None
try:
fd = os.open(NAME, os.O_RDONLY | os.O_NOFOLLOW, dir_fd=directory)
except OSError:
return None
finally:
os.close(directory)
with os.fdopen(fd, 'rb') as source:
info = os.fstat(source.fileno())
if not stat.S_ISREG(info.st_mode) or info.st_size > LIMIT:
return None
try:
value = json.loads(source.read(LIMIT))
except ValueError:
return None
return value if valid_copy(value) else None
def write_copy(root, value, owner):
directory = _directory(root, owner)
temporary = f'.{NAME}.{os.getpid()}'
try:
with contextlib.suppress(FileNotFoundError):
os.unlink(temporary, dir_fd=directory)
fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600, dir_fd=directory)
try:
with os.fdopen(fd, 'w') as output:
json.dump(value, output)
output.flush()
os.fchown(output.fileno(), *owner)
os.fsync(output.fileno())
os.rename(temporary, NAME, src_dir_fd=directory, dst_dir_fd=directory)
except BaseException:
with contextlib.suppress(OSError):
os.unlink(temporary, dir_fd=directory)
raise
os.fsync(directory)
finally:
os.close(directory)
def rclone_mount(config):
"""Where the hookscript of an Rclone container publishes its mount on the
host. The hookscript itself is written again from the engine."""
hook = RCLONE_HOOK.search(config)
if not hook:
return None
path = SNIPPETS / hook[1]
try:
if path.is_symlink():
return None
text = path.read_text()
except (OSError, UnicodeDecodeError):
return None
name = re.search(r'^inside=/data/mounts/(\S+)$', text, re.MULTILINE)
views = {key: re.search(rf'^{key}=(/\S+)/(\S+)$', text, re.MULTILINE) for key in ('published', 'published_ro')}
parent = re.search(r'^\s*if ! mountpoint -q (/\S+); then$', text, re.MULTILINE)
if not name or not parent or not all(views.values()) \
or any(view[2] != name[1] for view in views.values()):
return None
return {'mount_name': name[1], 'shared_mount_root': views['published'][1],
'shared_mount_read_only_root': views['published_ro'][1], 'shared_mount_root_parent': parent[1]}
def bundle(record, config):
"""What the container carries: its record and the host files a start of
the application depends on."""
vmid = record['vmid']
value = {'schema_version': 1, 'kind': KIND, 'node': socket.gethostname().split('.', 1)[0],
'vmid': vmid, 'record': record}
contract = Path(STACK_CONTRACT.format(vmid))
if contract.is_file() and not contract.is_symlink():
with contextlib.suppress(OSError, ValueError):
value['stack_contract'] = json.loads(contract.read_text())
hook = NVIDIA_HOOK.search(config)
if hook:
path = Path(hook[0])
with contextlib.suppress(OSError, UnicodeDecodeError):
if path.is_file() and not path.is_symlink() and sha(path.read_bytes()) == hook[1]:
value['nvidia_hook'] = {'sha256': hook[1], 'content': path.read_text()}
mount = rclone_mount(config)
if mount:
value['rclone_mount'] = mount
return value
def digest(value):
stable = {key: item for key, item in value.items() if key not in ('saved_at', 'generation')}
return sha(json.dumps(stable, sort_keys=True, separators=(',', ':')).encode())
def read_stamp(path):
"""What this host last wrote into the container: the digest of the copy
and the mark that tells that copy from any other."""
try:
text = path.read_text().strip()
except OSError:
return {}
try:
value = json.loads(text)
except ValueError:
return {'digest': text}
return value if isinstance(value, dict) else {}
def superseded(record, existing, stamp):
"""Whether the container carries a copy this host did not write, with a
record that differs from the one of this host. That is a container that
came back: migrated to another node and changed there, restored from a
backup made before the last operation, or rolled back to a snapshot. Its
content is the one its own copy describes."""
if not stamp.get('generation') or existing.get('generation') == stamp['generation']:
return False
carried_record = existing['record']
return (carried_record.get('installation_id') == record['installation_id']
and carried_record.get('vmid') == record['vmid'] and carried_record != record)
def carry(root, vmid, mount_stopped=True, verify=False, adopted=False):
"""Leave the current record inside its container. Returns 'carried',
'current' when the copy was already up to date, 'skipped' when the
instance is in the middle of an operation or cannot be reached, or
'stale' when the container came back with another record: the record of
this host is set aside, and the container is then recovered like any
restored one. A stopped container is opened only when its copy is known to
be old, or when `verify` asks to look anyway before an operation. A record
that was just recovered is `adopted`: it replaces the copies it came from."""
try:
record = instances.read(root, vmid)
except (OSError, ValueError, KeyError):
return 'skipped'
if record.get('status') != 'installed' or record.get('pending_transaction') \
or record.get('pending_stack_transaction'):
return 'skipped'
result = _run('pct', 'config', str(vmid))
if result.returncode != 0 or instances.identity(result.stdout.encode()) != record['installation_id']:
return 'skipped'
value = bundle(record, result.stdout)
expected = digest(value)
record_path = instances.location(root, vmid)
path = record_path.parent / STAMP
stamp = read_stamp(path)
if not adopted and replaced_elsewhere(record, record_path, read_cluster(vmid), value['node']):
record_path.replace(record_path.with_name(f"retired-{record['installation_id']}.json"))
with contextlib.suppress(OSError):
path.unlink()
return 'stale'
if not running_pid(vmid):
if stamp.get('digest') == expected and stamp.get('generation') and not verify:
keep_cluster(vmid, value, stamp['generation'])
return 'current'
if not mount_stopped:
return 'skipped'
try:
with container_root(vmid, mount_stopped) as rootfs:
if rootfs is None:
return 'skipped'
existing = read_copy(rootfs)
if existing is not None and not adopted and superseded(record, existing, stamp):
outcome = 'stale'
elif existing is not None and digest(existing) == expected and existing.get('generation') \
and existing.get('generation') == stamp.get('generation'):
outcome = 'current'
else:
value['saved_at'] = datetime.datetime.now(datetime.timezone.utc).isoformat()
value['generation'] = uuid.uuid4().hex
write_copy(rootfs, value, mapped_root(result.stdout))
stamp = {'generation': value['generation']}
outcome = 'carried'
if outcome == 'stale':
record_path.replace(record_path.with_name(f"retired-{record['installation_id']}.json"))
with contextlib.suppress(OSError):
path.unlink()
return outcome
path.write_text(json.dumps({'digest': expected, 'generation': stamp.get('generation')}) + '\n')
path.chmod(0o600)
if stamp.get('generation'):
keep_cluster(vmid, value, stamp['generation'])
except OSError:
return 'skipped'
return outcome
def registered(root):
if not root.is_dir():
return []
return sorted(int(d.name) for d in root.iterdir()
if d.name.isdecimal() and instances.has_contract(root, int(d.name)))
def sync(root, vmids=None, mount_stopped=True, verify=False):
"""Bring the copies of the given instances, or of all, up to date. Never
waits for the registry: an operation in progress carries its own copy
when it ends."""
try:
with instances.locked(root):
return {vmid: carry(root, vmid, mount_stopped, verify)
for vmid in (registered(root) if vmids is None else vmids)}
except (BlockingIOError, OSError, ValueError):
return {}
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('vmid', type=int, nargs='*')
parser.add_argument('--root', type=Path, default=instances.ROOT)
parser.add_argument('--running-only', action='store_true')
parser.add_argument('--verify', action='store_true')
args = parser.parse_args()
if os.geteuid() != 0:
parser.error(translate('Root privileges are required'))
print(json.dumps(sync(args.root, args.vmid or None, not args.running_only, args.verify)))
return 0
if __name__ == '__main__':
sys.exit(main())
+39
View File
@@ -114,9 +114,41 @@ configure_immich_ml_gpu() {
esac
}
# The generation of the AMD GPU as the compute driver names it, e.g. gfx1030.
amd_gfx_arch() {
local target
target=$(awk '$1 == "gfx_target_version" && $2 > 0 {print $2}' \
/sys/class/kfd/kfd/topology/nodes/*/properties 2>/dev/null | sort -n | tail -1)
[[ -n $target ]] || return 1
printf 'gfx%d%d%x' $((target / 10000)) $((target / 100 % 100)) $((target % 100))
}
validate_immich_ml_runtime() {
[[ $ML_ACCELERATION != cpu ]] || return 0
msg_info "$(translate "Checking the GPU of the machine learning container...")"
if [[ $ML_ACCELERATION == rocm ]]; then
# The provider being present says nothing about this GPU: the image
# carries kernels for some generations, and on any other every inference
# aborts.
local arch
if [[ -n ${ML_GFX_OVERRIDE:-} ]]; then
# ROCm is told to treat this GPU as the generation of its family.
arch="gfx${ML_GFX_OVERRIDE//./}"
else
arch=$(amd_gfx_arch) || arch=""
fi
if [[ -n $arch ]]; then
# Listed first: a match ends grep early, and with pipefail the listing
# it cut short would read as a failure.
local kernels
kernels=$(pct exec "$ML_ID" -- sh -c 'ls /opt/rocm/lib/rocblas/library 2>/dev/null' || true)
grep -Eq "[_-]${arch}\\.(dat|co|hsaco)" <<<"$kernels" || {
oci_log "The ROCm image has no kernels for $arch"
msg_warn "$(translate "The ROCm image has no support for the AMD GPU of this host:") $arch"
return 1
}
fi
fi
oci_quiet pct exec "$ML_ID" -- python -c '
import ctypes
import sys
@@ -135,5 +167,12 @@ else:
assert driver.cuInit(0) == 0, "CUDA driver initialization failed"
print("Immich ML GPU runtime:", profile, "available; model inference is tested separately")
' "$ML_ACCELERATION" || return
if [[ $ML_ACCELERATION == rocm ]]; then
# One real inference on the GPU: the provider alone does not prove it.
python3 "${SCRIPT_DIR}/oci_rocm_check.py" "$ML_ID" >>"${OCI_LOG:-/dev/null}" 2>&1 || {
msg_warn "$(translate "The AMD GPU did not complete a test inference with ROCm")"
return 1
}
fi
msg_ok "$(translate "GPU available for machine learning:") $ML_ACCELERATION"
}
+307
View File
@@ -0,0 +1,307 @@
#!/usr/bin/env python3
"""Change what runs the recognition of an installed Immich: the CPU or a GPU.
The machine learning container takes the image built for the new choice and
the devices that choice needs. Its model cache, the library, the database and
the other containers keep their data. The change is applied with an update of
the application, so a failure restores the previous containers, and with them
the previous choice.
"""
from __future__ import annotations
import argparse
import os
from pathlib import Path
import re
import subprocess
import sys
import oci_instances as instances
import oci_stack_modify
import oci_stack_native
from oci_ui import msg_error, msg_info, msg_ok, msg_warn, translate
ADAPTER = 'install_immich_stack.sh'
ENGINE = Path(__file__).resolve().parent
PROFILES = ('cpu', 'openvino', 'cuda', 'rocm')
RENDER = re.compile(r'/dev/dri/renderD[0-9]+')
# What a previous choice left in the container: its devices and the settings
# of its runtime.
GPU_DEVICE = re.compile(r'/dev/dri/renderD[0-9]+|/dev/kfd|/dev/nvidia\S*')
GPU_LINE = re.compile(r'lxc\.environment\.runtime: HSA_[A-Z_]+=|lxc\.environment: NVIDIA_[A-Z_]+=|'
r'lxc\.hook\.mount: /usr/local/lib/proxmenux/oci/nvidia-mount-')
ROCM_ROOTFS_GB = 40
NVIDIA_CAPABILITIES = 'compute,utility'
GPU_VARIABLES = ('NVIDIA_DRIVER_CAPABILITIES', 'HSA_OVERRIDE_GFX_VERSION', 'HSA_USE_SVM')
def run(*command):
result = subprocess.run(command, capture_output=True, text=True, check=False)
if result.returncode != 0:
detail = (result.stderr or result.stdout).strip().splitlines()[-1:] or ['']
raise RuntimeError(f"{' '.join(command[:3])}: {detail[0]}")
return result.stdout
def members(root, primary_id):
"""The record of each container of the application, by its role."""
primary = instances.read(root, primary_id)
found = {}
for member in (primary.get('stack') or {}).get('members', []):
record = instances.read(root, member['vmid'])
profile = record['deployment'].get('replay_profile') or {}
if profile.get('adapter') != ADAPTER:
raise ValueError(translate('This application is not an Immich installed by ProxMenux'))
found[profile.get('role')] = record
if set(found) != {'server', 'database', 'valkey', 'machine-learning'} or found['server']['vmid'] != primary_id:
raise ValueError(translate('This application is not an Immich installed by ProxMenux'))
return primary, found
def current(record):
return (record['deployment'].get('machine_learning') or {}).get('acceleration', 'cpu')
def conf(vmid):
return Path(f'/etc/pve/lxc/{int(vmid)}.conf')
def settings(vmid):
"""The current configuration of the container, one value per key."""
lines = [line.partition(': ') for line in run('pct', 'config', str(vmid)).splitlines()]
return {key: value for key, separator, value in lines if separator}
def strip_gpu(vmid):
"""Take the devices and runtime settings of the previous choice away."""
for key, value in oci_stack_modify.entries(vmid, 'dev').items():
if GPU_DEVICE.fullmatch(oci_stack_modify.device_path(value)):
run('pct', 'set', str(vmid), '--delete', key)
text = conf(vmid).read_text()
head, separator, snapshots = text.partition('\n[')
rebuilt = '\n'.join(line for line in head.split('\n') if not GPU_LINE.match(line)) + separator + snapshots
if rebuilt != text:
conf(vmid).write_text(rebuilt)
def append_lines(vmid, lines):
"""Add lines to the current configuration, before any snapshot section."""
head, separator, snapshots = conf(vmid).read_text().partition('\n[')
conf(vmid).write_text(head.rstrip('\n') + '\n' + '\n'.join(lines) + '\n' + separator + snapshots)
def add_device(vmid, path):
info = os.stat(path)
run('pct', 'set', str(vmid), '--' + oci_stack_modify.free_key(vmid, 'dev'),
f'path={path},gid={info.st_gid},mode=0660')
def configure(vmid, acceleration, render_device, gfx_override):
"""Give the stopped container what the new choice needs."""
strip_gpu(vmid)
if acceleration in ('openvino', 'rocm'):
add_device(vmid, render_device)
if acceleration == 'rocm':
add_device(vmid, '/dev/kfd')
if gfx_override:
append_lines(vmid, [f'lxc.environment.runtime: HSA_OVERRIDE_GFX_VERSION={gfx_override}',
'lxc.environment.runtime: HSA_USE_SVM=0'])
size = re.search(r'size=([0-9]+)G', settings(vmid).get('rootfs', ''))
if size and int(size[1]) < ROCM_ROOTFS_GB:
# The ROCm image is several times larger than the others.
run('pct', 'resize', str(vmid), 'rootfs', f'{ROCM_ROOTFS_GB}G')
if acceleration == 'cuda':
# The same NVIDIA setup the installation uses.
run('bash', '-c', f'set -e; SCRIPT_DIR="{ENGINE}"; source "$SCRIPT_DIR/oci_ui.sh"; '
'die() { msg_error "$*"; exit 1; }; oci_log() { :; }; '
'source "$SCRIPT_DIR/oci_nvidia_setup.sh"; source "$SCRIPT_DIR/oci_immich_ml.sh"; '
f'configure_immich_nvidia {int(vmid)} {NVIDIA_CAPABILITIES}')
run('pct', 'set', str(vmid), '--memory', str(memory(acceleration)))
values = settings(vmid)
limited = 'cpulimit' in values and 'cores' not in values
# Intel keeps the CPU topology and limits its share of time; the others
# are given four cores.
if acceleration == 'openvino' and not limited:
run('pct', 'set', str(vmid), '--cpulimit', '4', '--delete', 'cores')
elif acceleration != 'openvino' and limited:
run('pct', 'set', str(vmid), '--cores', '4', '--delete', 'cpulimit')
def validate(acceleration, render_device, gfx_override):
if acceleration not in PROFILES:
raise ValueError(translate('Immich GPU profile not validated'))
if acceleration in ('openvino', 'rocm'):
if not render_device or not RENDER.fullmatch(render_device) or not Path(render_device).is_char_device():
raise ValueError(f"{translate('The render device does not exist:')} {render_device}")
vendor = Path(f'/sys/class/drm/{Path(render_device).name}/device/vendor').read_text().strip()
if vendor != ('0x8086' if acceleration == 'openvino' else '0x1002'):
raise ValueError(translate('The render device does not belong to the GPU of that choice'))
if acceleration == 'rocm' and not Path('/dev/kfd').is_char_device():
raise ValueError(translate('ROCm requires /dev/kfd on the host'))
if acceleration == 'cuda' and subprocess.run(['nvidia-smi', '-L'], capture_output=True, check=False).returncode:
raise ValueError(translate('The host NVIDIA driver is not responding correctly'))
if gfx_override and (acceleration != 'rocm' or not re.fullmatch(r'[0-9]{1,2}\.[0-9]\.[0-9]', gfx_override)):
raise ValueError(translate('Invalid ROCm generation override'))
def recorded(record, acceleration, render_device, gfx_override):
"""The recognition settings of a record after the change."""
settings = dict(record['deployment'].get('machine_learning') or {})
settings.update(acceleration=acceleration,
render_device=render_device if acceleration in ('openvino', 'rocm') else None,
gfx_override=gfx_override if acceleration == 'rocm' else None)
settings['resources'] = dict(settings.get('resources') or {}, memory_mb=memory(acceleration),
cpu_allocation='quota' if acceleration == 'openvino' else 'cpuset')
return settings
def memory(acceleration):
return 4096 if acceleration == 'cpu' else 8192
def resources(current, acceleration):
"""The resources a record declares for the container of the new choice."""
result = dict(current, memory_mb=memory(acceleration))
result.pop('cpu_allocation', None)
if acceleration == 'openvino':
result['cpu_allocation'] = 'quota'
return result
def declared(environment, acceleration, gfx_override):
"""The variables a record declares, with the ones of the new choice."""
result = [entry for entry in environment if entry.get('name') not in GPU_VARIABLES]
if acceleration == 'cuda':
result.append({'name': 'NVIDIA_DRIVER_CAPABILITIES', 'value': NVIDIA_CAPABILITIES})
if acceleration == 'rocm' and gfx_override:
result += [{'name': 'HSA_OVERRIDE_GFX_VERSION', 'value': gfx_override}, {'name': 'HSA_USE_SVM', 'value': '0'}]
return result
def declare(record, acceleration, gfx_override):
"""A record an update already rebuilt describes its container by itself:
it takes the variables, the resources and the disk of the new choice."""
plan = record['deployment']
if 'native_config' in plan:
return
plan['environment'] = declared(plan.get('environment', []), acceleration, gfx_override)
plan['resources'] = resources(plan.get('resources') or {}, acceleration)
size = re.search(r'size=([0-9]+)G', settings(record['vmid']).get('rootfs', ''))
if size and isinstance(plan.get('rootfs'), dict):
plan['rootfs']['size_gb'] = int(size[1])
profile = (record['template'].get('proxmox') or {}).get('installer_profile')
if isinstance(profile, dict):
profile.pop('cpu_allocation', None)
if acceleration == 'openvino':
profile['cpu_allocation'] = 'quota'
def chosen(record):
"""What runs recognition according to a record."""
saved = record['deployment'].get('machine_learning') or {}
return saved.get('acceleration', 'cpu'), saved.get('render_device'), saved.get('gfx_override')
def stop(vmid):
try:
run('pct', 'shutdown', str(vmid), '--timeout', '60')
except RuntimeError:
run('pct', 'stop', str(vmid))
def apply(root, primary_id, vmid, images, acceleration, render_device, gfx_override):
"""Give the stopped container and the records one choice."""
configure(vmid, acceleration, render_device, gfx_override)
learning = instances.read(root, vmid)
learning['template']['container_contract']['image']['reference'] = images[acceleration]
learning['deployment']['machine_learning'] = recorded(learning, acceleration, render_device, gfx_override)
declare(learning, acceleration, gfx_override)
instances.write(instances.location(root, vmid), learning)
primary = instances.read(root, primary_id)
primary['stack']['deployment']['machine_learning'] = recorded(
{'deployment': primary['stack']['deployment']}, acceleration, render_device, gfx_override)
intent = (primary.get('native_stack_intent') or {}).get('deployment')
if isinstance(intent, dict) and 'machine_learning' in intent:
intent['machine_learning'] = recorded({'deployment': intent}, acceleration, render_device, gfx_override)
instances.write(instances.location(root, primary_id), primary)
oci_stack_modify.register(root, vmid)
def revert(root, primary_id, vmid, images, previous, was_running):
"""Give the container and the records the choice they had."""
try:
if oci_stack_modify.is_running(vmid):
stop(vmid)
apply(root, primary_id, vmid, images, *previous)
except (OSError, ValueError, KeyError, RuntimeError, subprocess.SubprocessError) as error:
msg_warn(f"{translate('The previous recognition choice could not be put back:')} {error}")
return
if was_running:
subprocess.run(['pct', 'start', str(vmid)], capture_output=True, check=False)
def change(root, primary_id, acceleration, render_device=None, gfx_override=None):
validate(acceleration, render_device, gfx_override)
with instances.locked(root):
primary, found = members(root, primary_id)
learning = found['machine-learning']
vmid = learning['vmid']
if any(record.get('status') != 'installed' or record.get('pending_transaction')
or record.get('pending_stack_transaction') for record in found.values()):
raise ValueError(translate('The container has an operation pending; finish or recover it first'))
if current(learning) == acceleration:
raise ValueError(translate('Recognition already runs on that choice'))
images = primary['stack']['template']['proxmox']['application_options']['machine_learning']['profile_images']
previous = chosen(learning)
was_running = oci_stack_modify.is_running(vmid)
try:
if was_running:
msg_info(translate('Stopping the container...'))
stop(vmid)
msg_ok(translate('Container stopped'))
msg_info(translate('Preparing the machine learning container for the new choice...'))
apply(root, primary_id, vmid, images, acceleration, render_device, gfx_override)
msg_ok(translate('Machine learning container prepared'))
except BaseException:
revert(root, primary_id, vmid, images, previous, was_running)
raise
try:
# The image of the new choice replaces the one in use, as an update does.
oci_stack_native.run(primary_id, acknowledge_external_data=True)
except BaseException:
with instances.locked(root):
# An update left halfway keeps its own record of what to restore.
if not instances.read(root, primary_id).get('pending_stack_transaction'):
revert(root, primary_id, vmid, images, previous, was_running)
raise
if was_running and not oci_stack_modify.is_running(vmid):
# An update leaves each container as it found it, and this one was
# stopped for the change.
run('pct', 'start', str(vmid))
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('vmid', type=int, help='main container of the application')
parser.add_argument('--acceleration', required=True, choices=PROFILES)
parser.add_argument('--render-device')
parser.add_argument('--gfx-override')
parser.add_argument('--root', type=Path, default=instances.ROOT)
args = parser.parse_args()
if os.geteuid() != 0:
parser.error(translate('Root privileges are required'))
try:
change(args.root, args.vmid, args.acceleration, args.render_device, args.gfx_override)
except BlockingIOError:
msg_error(translate('Another OCI operation is using the instance registry. Wait for it to finish.'))
return 1
except (OSError, ValueError, KeyError, RuntimeError, StopIteration, subprocess.SubprocessError) as error:
if not getattr(error, 'oci_reported', False):
msg_error(f"{translate('The recognition of Immich was not changed:')} {error}")
return 1
msg_ok(translate('Recognition of Immich changed; the application was updated with the new choice.'))
return 0
if __name__ == '__main__':
sys.exit(main())
+16 -2
View File
@@ -88,6 +88,8 @@ def image_from_archive(path):
manifest = blob(digest)
config = blob(manifest['config']['digest'])
return {'manifest_digest': digest, 'config_digest': manifest['config']['digest'],
'layers': [layer['digest'] for layer in manifest.get('layers', [])],
'created': config.get('created'),
'architecture': config['architecture'], 'os': config.get('os'),
'defaults': {k: config.get('config', {}).get(k) for k in FIELDS}}
@@ -190,8 +192,20 @@ def resolve_candidate(reference, architecture):
# The build date identifies the image as the publisher released it: it is
# what changes when an image is rebuilt, whether or not the application
# version inside it moved.
return {'manifest_digest': digest, 'defaults': defaults, 'version': version,
'created': config.get('created')}
return {'manifest_digest': digest, 'layers': [layer['digest'] for layer in json.loads(raw).get('layers', [])],
'defaults': defaults, 'version': version, 'created': config.get('created')}
def same_image(candidate, image):
"""Whether what the registry serves is the image of an archive or a record.
The OCI archive made from a manifest published in the Docker format has
another manifest and another configuration digest: both are rewritten. Its
layers and the moment the image was built are the same in both."""
if candidate['manifest_digest'] == image.get('manifest_digest'):
return True
return bool(candidate.get('layers') and candidate.get('created')
and candidate['layers'] == image.get('layers') and candidate['created'] == image.get('created'))
def compare(record, current, candidate=None):
+20 -4
View File
@@ -764,9 +764,11 @@ def restore_description(vmid, state):
run('pct', 'set', str(vmid), '--description', original_description(state))
def release_stage(state):
def release_stage(state, discard=False):
"""After a commit the holder CT only keeps its own rootfs: every parked
volume went back to the application. Anything still attached keeps it."""
volume went back to the application. After a verified recovery it holds
the disks of the failed attempt, which the restored backup replaced:
`discard` removes them with it. Otherwise anything still attached keeps it."""
stage = state.get('stage')
if not stage:
return
@@ -774,11 +776,23 @@ def release_stage(state):
config = owned(stage, 'proxmenux-transaction=' + state['id'])
except (ValueError, RuntimeError, subprocess.CalledProcessError):
return
if mounts(config) or any(re.fullmatch(r'unused[0-9]+', key) for key in parse_config(config)):
held = mounts(config) or any(re.fullmatch(r'unused[0-9]+', key) for key in parse_config(config))
if held and not discard:
return
if discard:
stop(stage)
run('pct', 'destroy', str(stage))
def discard_stage(state):
"""Remove the holder CT of an operation that ended in a verified recovery.
A failure here leaves it in place and does not undo the recovery."""
try:
release_stage(state, discard=True)
except (OSError, ValueError, RuntimeError, subprocess.SubprocessError) as error:
log(f'cleanup: {error}')
def gib(size):
return f'{size / 1024**3:.1f} GB'
@@ -1124,7 +1138,9 @@ def complete_recovery(root, journal, state):
instances.write(instances.location(root, vmid), restored)
checkpoint(journal, state, 'rolled-back')
if show:
msg_ok(translate('Recovery completed. The displaced disks and the backup are kept; nothing was deleted automatically.'))
# A member of a stack is released when the whole stack is back.
discard_stage(state)
msg_ok(translate('Recovery completed. The disks of the failed attempt were removed; the backup is kept.'))
if state.get('original_host_sources') or state.get('desired_host_sources'):
msg_info2(translate('Shared host files are kept as they are; the backup does not restore their content.'))
+37 -4
View File
@@ -2,12 +2,14 @@
"""Instance recording adapter for the existing dedicated stack installers."""
import argparse
import copy
import ipaddress
import json
import os
from pathlib import Path
import re
import subprocess
import sys
import time
import oci_console
import oci_instances as instances
@@ -103,6 +105,40 @@ def capture_rootfs(root, vmid):
instances.write(instances.location(root, vmid), record)
PRIVATE_STACK_NETWORK = ipaddress.ip_network('10.77.0.0/16')
def access_address(vmid, wait=20):
"""The address a container is reached at from the local network. A member
of a multi-container application also has a leg on its private network,
and that address answers from the host only: it is used when the
container has no other."""
def addresses():
result = subprocess.run(['lxc-info', '-n', str(vmid), '-iH'], capture_output=True, text=True, timeout=5)
found = []
for line in result.stdout.splitlines():
try:
found.append(ipaddress.IPv4Address(line.strip()))
except ValueError:
continue
return found
def outside(found):
return next((str(address) for address in found if address not in PRIVATE_STACK_NETWORK), '')
found = addresses()
config = subprocess.run(['pct', 'config', str(vmid)], capture_output=True, text=True, timeout=30).stdout
legs = re.findall(r'^net[0-9]+: .*?(?:^|,)ip=([^,\s]+)', config, re.MULTILINE)
# A leg outside the private network may still be waiting for its lease.
expected = any(leg == 'dhcp' or (leg[:1].isdigit() and ipaddress.ip_interface(leg).ip not in PRIVATE_STACK_NETWORK)
for leg in legs)
deadline = time.monotonic() + (wait if expected else 0)
while not outside(found) and time.monotonic() < deadline:
time.sleep(2)
found = addresses()
return outside(found) or (str(found[0]) if found else '')
def finalize(root, primary):
parent = instances.read(root, primary)
intent = parent['native_stack_intent']
@@ -118,10 +154,7 @@ def finalize(root, primary):
presentation['catalog_ui'] = {**stack_ui, **(presentation.get('catalog_ui') or {})}
if not (presentation.get('first_run') or {}).get('endpoints'):
presentation['first_run'] = intent['template'].get('first_run') or {}
ip_result = subprocess.run(['lxc-info', '-n', str(vmid), '-iH'],
capture_output=True, text=True, timeout=5)
ip = next((line.strip() for line in ip_result.stdout.splitlines()
if re.fullmatch(r'[0-9]+(?:\.[0-9]+){3}', line.strip())), '')
ip = access_address(vmid)
description = render(presentation, plan['image']['manifest_digest'],
record['installation_id'], ip)
subprocess.run(['pct', 'set', str(vmid), '--description', description], check=True)
+30
View File
@@ -68,6 +68,36 @@ def create_parent(path, root):
os.chown(path, owner.st_uid, owner.st_gid)
def rebuild(root, vmid, same_gpu=False):
"""Point a stopped container at the NVIDIA driver of this host. The caller
holds the registry and the container is not started: it is the step a
restore on a host with another driver needs before the first start.
Returns whether anything had to change."""
record = instances.read(root, vmid)
config = instances.command('pct', 'config', str(vmid))
if not instances.same_config_except_notes(record, config):
raise ValueError(translate('The container identity or configuration changed'))
plan = nv.refresh_plan(config, record['observed']['gpu_devices'][nv.KEY], same_gpu=same_gpu)
if not plan['changed']:
return False
if instances.command('pct', 'status', str(vmid)).strip() != b'status: stopped':
raise ValueError(translate('Stop the container before the NVIDIA refresh'))
instances.command('pct', 'mount', str(vmid))
try:
candidate = prepare(Path(f'/var/lib/lxc/{vmid}/rootfs'), plan)
finally:
instances.command('pct', 'unmount', str(vmid))
Path(f'/etc/pve/lxc/{vmid}.conf').write_bytes(candidate)
updated = copy.deepcopy(record)
updated['observed'] = instances.observe(vmid, record['installation_id'],
record['observed']['archive_path'], record['observed']['resolved_registry_digest'],
record['observed']['image'])
nv.check_mounts(updated['observed']['config'].encode(), plan['inventory'])
nv.check_devices(updated['observed']['config'].encode(), plan['inventory'])
instances.write(instances.location(root, vmid), updated)
return True
def refresh(root, vmid, apply=False):
with instances.locked(root):
record = instances.read(root, vmid)
+4 -2
View File
@@ -75,11 +75,13 @@ def verify(value):
raise ValueError(translate('The NVIDIA driver or inventory changed; the operation was stopped'))
def refresh_plan(config, previous, current=None):
def refresh_plan(config, previous, current=None, same_gpu=True):
"""Resolve current host components without treating a driver version as intent.
This only prepares a plan; applying it requires a stopped-CT transaction and
preparing file destinations/library links before the next native start.
A container restored on another host takes the GPU of that host: the
application asked for the NVIDIA runtime, not for one card.
"""
check_devices(config, previous)
check_mounts(config, previous)
@@ -92,7 +94,7 @@ def refresh_plan(config, previous, current=None):
raise ValueError(translate('Incomplete NVIDIA identity'))
result.append(tuple(fields[:2]))
return sorted(result)
if identities(previous) != identities(current):
if same_gpu and identities(previous) != identities(current):
raise ValueError(translate('The physical NVIDIA selection changed'))
# Remove only entries already validated against our recorded inventory.
kept = []
+5 -1
View File
@@ -29,6 +29,7 @@ CLUSTER_NODES = Path('/etc/pve/nodes')
SNIPPETS = Path('/var/lib/vz/snippets')
# The App tab of ProxMenux Monitor keeps one file per VMID.
MONITOR_APPS = Path('/etc/proxmenux/apps')
CLUSTER_RECORDS = Path('/etc/pve/priv/proxmenux/oci')
HOST_MONITOR_INCLUDES = (Path('/etc/pve/proxmenux/host-monitor'), Path('/etc/pve/lxc/proxmenux-host-monitor'))
STACK_HOOK = 'proxmenux-stack-dependencies.sh'
@@ -136,6 +137,7 @@ def remove_host_state(vmid):
except OSError:
pass
_unlink(hook)
_unlink(CLUSTER_RECORDS / f'{int(vmid)}.json')
_unlink(MONITOR_APPS / f'{int(vmid)}.json')
dismissed = MONITOR_APPS / '.oci-dismissed.json'
try:
@@ -178,6 +180,7 @@ def _leftovers(vmid):
"""Whether anything of the container is still on the host."""
paths = [runtime_settings.include_path(vmid), runtime_settings.legacy_include_path(vmid),
SNIPPETS / f'proxmenux-rclone-{int(vmid)}-fuse-hook.sh', MONITOR_APPS / f'{int(vmid)}.json',
CLUSTER_RECORDS / f'{int(vmid)}.json',
*oci_console.LOG_DIR.glob(f'{int(vmid)}.console.log*')]
return any(path.exists() for path in paths)
@@ -191,7 +194,8 @@ def sweep_orphans(root):
for directory, pattern in ((runtime_settings.include_path(0).parent, r'([0-9]+)\.sysctls'),
(runtime_settings.legacy_include_path(0).parent, r'([0-9]+)\.proxmenux-sysctls'),
(oci_console.LOG_DIR, r'([0-9]+)\.console\.log.*'),
(SNIPPETS, r'proxmenux-rclone-([0-9]+)-fuse-hook\.sh')):
(SNIPPETS, r'proxmenux-rclone-([0-9]+)-fuse-hook\.sh'),
(CLUSTER_RECORDS, r'([0-9]+)\.json')):
if directory.is_dir():
found.update(int(m.group(1)) for m in (re.fullmatch(pattern, p.name) for p in directory.iterdir()) if m)
# Only the App tab registrations of OCI installs; the other ones belong to
File diff suppressed because it is too large Load Diff
+81
View File
@@ -0,0 +1,81 @@
#!/usr/bin/env python3
"""A real inference on the AMD GPU of a container, through ROCm.
That the MIGraphX provider is listed says nothing about the GPU of this host:
an image carries kernels for some generations only, and on any other the
first inference aborts. This runs a small convolution model on the GPU inside
the container and fails when it does not come back with a result.
"""
from __future__ import annotations
import argparse
import re
import subprocess
import sys
# A model of five operators (convolution, pooling and a dense layer), enough
# to make ROCm compile and run on the GPU without downloading anything.
MODEL = (
'CAg65wUKPAoBeAoBdwoBYhIBYyIEQ29udioVCgxrZXJuZWxfc2hhcGVAA0ADoAEHKhEKBHBhZHNAAUABQAFAAaABBwoMCgFjEgFy'
'IgRSZWx1ChkKAXISAWciEUdsb2JhbEF2ZXJhZ2VQb29sCg8KAWcSAWYiB0ZsYXR0ZW4KFAoBZgoCZncKAmZiEgF5IgRHZW1tEgVw'
'cm9iZSrAAwgECAMIAwgDEAFCAXdKsAOu/QA5e7v0POCS4LyqZLa9sDs6vdcWy70cFMU78DwJPpibSb2CJX69qqNIPVEuEj3xtSw8'
'U4++vWu0P7vqZY49x6UJvn5wO71qr0K+dgwEvvuXPL4vlsC8WskBvkI43jwWaYA8QyKZvKzbgL4Lply9i+2euzylOTyXrxy+ELBD'
'vZVmyL1epqW9pUXZPRNipb1fIlW7gB+1PfIKb70yAze8Bfw0PAgA0Ts15Pq9D3/5O74kCz54bR6+ZwCwPbWMQzyGX4O9uNdMPl0c'
'nD1HnfW9vyz0O0o2bD17ppq8K9yLPcf22bv9pog9Ak4TPipgir1BaaY8U8U9vT+EUDwwI/O9LUhtvUe5oLwdEbg9nYrqPX2HB74m'
'vqK9X3yEPRcGTL7itj29GGUfvOW3AD6fMI090AYGvfz3Fr3I9cy8aQIcPqtRL71lxvi8pWsQPczeRbyAnaG8NSnkvaIDl7rdsDW9'
'r9LuPabAhT1DOh67a+KIPeg1C726edc914sNuhP0bj3+LwS+CgAOPUPfLL7talC+b235vCBOuL1eZIY889xlPj9Wqr06kX+9VESo'
'PDHwST0qGQgEEAFCAWJKEAAAAAAAAAAAAAAAAAAAAAAqLAgECAIQAUICZndKIAmDkLy5sqi8S92PPUX0VD1istO9DrsBvIJBZztd'
'9de9KhIIAhABQgJmYkoIAAAAAAAAAABaGwoBeBIWChQIARIQCgIIAQoCCAMKAggICgIICGITCgF5Eg4KDAgBEggKAggBCgIIAkIE'
'CgAQDQ=='
)
PROBE = """
import base64, sys
import numpy as np
import onnxruntime as ort
options = ort.SessionOptions()
options.log_severity_level = 3
session = ort.InferenceSession(base64.b64decode(sys.argv[1]), options, providers=['MIGraphXExecutionProvider'])
assert session.get_providers()[0] == 'MIGraphXExecutionProvider', session.get_providers()
result = session.run(None, {'x': np.ones((1, 3, 8, 8), np.float32)})[0]
assert result.shape == (1, 2) and np.isfinite(result).all(), result
print('ROCm inference on the GPU: ok')
"""
def variables(config):
"""What the container tells ROCm about its GPU. A command run in the
container does not receive the variables its application starts with."""
found = re.findall(r'^lxc\.environment\.runtime: (.+)$', config, re.M)
for line in re.findall(r'^env: (.+)$', config, re.M):
found += line.split('\0')
return [value for value in found if re.fullmatch(r'HSA_[A-Z_]+=[0-9A-Za-z.]+', value)]
def check(vmid, timeout=300):
"""Raises RuntimeError when the container cannot run the model on its GPU."""
vmid = str(int(vmid))
config = subprocess.run(['pct', 'config', vmid], capture_output=True, text=True, check=False).stdout
try:
result = subprocess.run(['pct', 'exec', vmid, '--', 'env', *variables(config), 'python', '-c', PROBE, MODEL],
capture_output=True, text=True, timeout=timeout, check=False)
except subprocess.TimeoutExpired as error:
raise RuntimeError('the GPU did not answer in time') from error
if result.returncode != 0:
detail = (result.stderr or result.stdout).strip().splitlines()[-1:] or ['']
raise RuntimeError(detail[0][:300])
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('vmid', type=int)
args = parser.parse_args()
try:
check(args.vmid)
except RuntimeError as error:
print(error, file=sys.stderr)
return 1
return 0
if __name__ == '__main__':
sys.exit(main())
+20 -11
View File
@@ -377,6 +377,8 @@ class NativeAdapter:
member_tx.run('pct', 'exec', str(vmid), '--', 'python', '-c',
'import onnxruntime as ort; '
'assert "MIGraphXExecutionProvider" in ort.get_available_providers()')
import oci_rocm_check
oci_rocm_check.check(vmid)
if acceleration in ('openvino', 'cuda'):
member_tx.run('pct', 'exec', str(vmid), '--', 'python', '-c',
'import sys,ctypes,onnxruntime as ort; p=sys.argv[1]; '
@@ -561,7 +563,7 @@ class NativeAdapter:
record.pop('pending_stack_transaction', None)
instances.write(instances.location(self.root, vmid), record)
try:
self.release_stages()
self.release_stages(recovered=state['phase'] == 'rolled-back')
if self.keep_backup and state['phase'] == 'committed':
import oci_keep_backup
for backup in sorted(self.journal.parent.glob('backup-*/vzdump-lxc-*.tar.zst')):
@@ -574,18 +576,25 @@ class NativeAdapter:
except (OSError, ValueError) as exc:
member_tx.log(f'cleanup: {exc}')
def release_stages(self):
"""The temporary containers that held the data of each member; one
that still has a disk attached is kept."""
def release_stages(self, recovered=False):
"""The temporary containers that held the data of each member. When
the whole stack was recovered they are removed with the disks of the
failed attempt; otherwise one that still has a disk attached is kept."""
journals = [(path, True) for path in self.journal.parent.glob('recovery-*/transaction.json')]
for vmid in self.records:
folder = instances.location(self.root, vmid).parent / 'transactions'
for member_journal in folder.glob('*/transaction.json'):
try:
state = json.loads(member_journal.read_text())
except (OSError, ValueError):
continue
if (state.get('coordinated') or {}).get('journal') == str(self.journal):
member_tx.release_stage(state)
journals += [(path, False) for path in folder.glob('*/transaction.json')]
for member_journal, own in journals:
try:
state = json.loads(member_journal.read_text())
except (OSError, ValueError):
continue
if not own and (state.get('coordinated') or {}).get('journal') != str(self.journal):
continue
if recovered and state.get('phase') == 'rolled-back':
member_tx.discard_stage(state)
else:
member_tx.release_stage(state)
def prune_backups(self, include_current):
"""The backups of closed operations are removed, those of this one
+4 -4
View File
@@ -12,7 +12,7 @@ import tempfile
import oci_instances as instances
import oci_instance_transaction as transaction
from oci_installation_state import image_from_archive
from oci_installation_state import image_from_archive, same_image
from oci_ui import translate, msg_info, msg_ok, msg_warn, msg_error, msg_info2
@@ -65,7 +65,7 @@ def resolve_archive(desired, config, current=None, check=None):
if not re.fullmatch(r'sha256:[a-f0-9]{64}', digest):
raise ValueError(translate('Invalid registry digest'))
msg_ok(f"{translate('Image:')} {reference} ({candidate.get('version') or digest[7:19]})")
if digest == current:
if current and same_image(candidate, current):
return None, digest
if check:
check()
@@ -85,7 +85,7 @@ def resolve_archive(desired, config, current=None, check=None):
translate('The image did not pass the integrity check'))
except RuntimeError:
return False
return image_from_archive(str(path))['manifest_digest'] == digest
return same_image(candidate, image_from_archive(str(path)))
if archive.exists():
msg_info(translate('Verifying the image integrity...'))
@@ -161,7 +161,7 @@ def update(vmid, acknowledge_external_data=False, proposal=None, keep_backup=Non
changes = transaction.external_changes(record, config)
transaction.preflight(record, desired, config)
msg_ok(translate('Container checked'))
current = record['observed']['image']['manifest_digest'] if operation == 'update' else None
current = record['observed']['image'] if operation == 'update' else None
archive, digest = resolve_archive(desired, config, current, lambda: transaction.require_backup_space(
instances.location(instances.ROOT, vmid).parent, [vmid]))
if archive is None:
+14 -1
View File
@@ -17,10 +17,23 @@ find_snippet_storage() {
fi
storage=$(pvesm status --content snippets 2>/dev/null \
| awk 'NR > 1 && $3 == "active" {print $1; exit}')
[[ -n $storage ]] || die "No active storage supports snippets"
if [[ -z $storage ]]; then
# A new Proxmox installation accepts no snippets on any storage.
enable_local_snippets || die "No active storage supports snippets"
storage=local
fi
printf '%s' "$storage"
}
enable_local_snippets() {
local content
content=$(pvesh get /storage/local --output-format json 2>/dev/null | jq -r '.content // empty')
[[ -n $content ]] || return 1
pvesm set local --content "${content},snippets" >/dev/null 2>&1 || return 1
pvesm status --content snippets 2>/dev/null \
| awk 'NR > 1 && $1 == "local" && $3 == "active" {found=1} END {exit !found}'
}
install_hook() {
local main_id=${1:?missing main VMID} source_config=${2:?missing lifecycle JSON}
local storage hook_volume hook_path target_config
+1 -2
View File
@@ -575,7 +575,6 @@ def _stack_storage(services: dict[str, Any], markers: dict[str, set[str]]) -> li
r"^/(?:data|downloads?|media|movies?|music|photos?|pictures?|recordings?|tv|videos?)(?:/|$)|/(?:library|uploads?)(?:/|$)",
re.I,
)
disposable_targets = {"/cache", "/tmp", "/transcode"}
system_targets = {"/etc/localtime", "/etc/timezone"}
runtime_targets = {"/var/run/docker.sock", "/run/docker.sock"}
for service_name, service in services.items():
@@ -609,7 +608,7 @@ def _stack_storage(services: dict[str, Any], markers: dict[str, set[str]]) -> li
"container_path": target,
"mode": mode,
"user_selectable": shareable,
"backup": mode == "managed-volume" and target not in disposable_targets,
"backup": mode == "managed-volume",
"shared_with_other_lxc": shareable,
"source_path": str(source) if system_bind else None,
"source_path_prompt": (
+11
View File
@@ -422,6 +422,8 @@ def install_template(ui, template: dict[str, Any], identifier: str, mode: str) -
console.wait_for_enter(translate("Press Enter to return to the menu..."))
return None
if result:
from .management import carry_records
carry_records(PROJECT_ROOT)
vmids = {int(v) for v in [result.get("vmid"), *(result.get("stack_vmids") or {}).values()] if v}
_, removed = images.offer_removal(ui, sorted(vmids))
_print_installation_summary(result, removed)
@@ -436,6 +438,8 @@ def _rclone_mount(catalog: Catalog, ui) -> None:
console.msg_title(translate("Rclone mount"))
result = run_remote_rclone_mount(PROJECT_ROOT, template, deployment, "auto")
if result:
from .management import carry_records
carry_records(PROJECT_ROOT)
console.msg_ok(f"Remote: {result['remote']}:")
console.msg_ok(f"{translate('Read/write')}: {result['read_write_path']}")
console.msg_ok(f"{translate('Read-only')}: {result['read_only_path']}")
@@ -590,6 +594,7 @@ def build_parser() -> argparse.ArgumentParser:
rclone_parser = subparsers.add_parser("rclone-mount", help="Enable a mount on an installed Rclone OCI")
rclone_parser.add_argument("--host", default="auto")
rclone_parser.add_argument("--dry-run", action="store_true")
subparsers.add_parser("recover", help="Register again the OCI applications restored from a backup")
manage_parser = subparsers.add_parser("manage", help="Update or recreate one installed OCI instance")
manage_parser.add_argument("vmid", type=int)
manage_parser.add_argument("--action", choices=("update", "recreate"), required=True)
@@ -654,8 +659,14 @@ def main(argv: list[str] | None = None) -> int:
return 0
result = run_remote_install(PROJECT_ROOT, template, deployment, args.host, args.dry_run)
if result:
if not args.dry_run:
from .management import carry_records
carry_records(PROJECT_ROOT)
_print_installation_summary(result)
return 0
if args.command == "recover":
from .management import direct_recovery
return direct_recovery(PROJECT_ROOT)
if args.command == "manage":
from .management import direct_management
lifecycle_args = []
+1 -1
View File
@@ -233,7 +233,7 @@ def _mount_contract(value: Any, optional: set[str]) -> list[dict[str, Any]]:
else "skip"
),
"managed_volume": {
"backup": target not in {"/cache", "/tmp", "/transcode"},
"backup": True,
"default_size_gb": 4 if target == "/config" else 8,
},
}
+79 -2
View File
@@ -141,7 +141,7 @@ def gpus(root: Path = Path("/")) -> dict[str, Any]:
"""The GPUs an installation can use: the render nodes of each Intel and
AMD GPU, and whether NVIDIA is usable on the host."""
vendors = {"0x8086": "intel", "0x1002": "amd"}
found: dict[str, Any] = {"intel": [], "amd": [], "nvidia": False}
found: dict[str, Any] = {"intel": [], "amd": [], "nvidia": False, "amd_gfx_target": None}
for node in sorted((root / "sys/class/drm").glob("renderD*")):
vendor = vendors.get(_sysfs(node / "device/vendor").lower())
if vendor:
@@ -153,14 +153,91 @@ def gpus(root: Path = Path("/")) -> dict[str, Any]:
timeout=15, check=False).returncode == 0
except (OSError, subprocess.TimeoutExpired):
found["nvidia"] = False
found["amd_gfx_target"] = amd_gfx_target(root)
return found
def amd_gfx_target(root: Path = Path("/")) -> int | None:
"""The generation of the AMD GPU as its compute driver reports it: 90012
for gfx90c (Vega integrated graphics), 100300 for gfx1030 (RDNA2). None
when the driver exposes no compute device."""
best = None
for path in sorted((root / "sys/class/kfd/kfd/topology/nodes").glob("*/properties")):
match = re.search(r"^gfx_target_version (\d+)$", _sysfs(path), re.MULTILINE)
if match and int(match[1]) > 0:
best = max(best or 0, int(match[1]))
return best
# The GPU generations the ROCm images of the catalog carry kernels for: RDNA2
# and newer, as gfx_target_version reports them, and the Instinct accelerators.
# On any other one the compute process of the image aborts.
ROCM_TARGETS = (100300, 110000, 110001, 110002, 110500, 110501, 120000, 120001)
ROCM_ACCELERATOR_TARGETS = (90008, 90010, 90402, 90500)
def gfx_name(target: int) -> str:
"""The name ROCm gives a generation: 90012 is gfx90c, 100300 is gfx1030."""
return f"gfx{target // 10000}{target // 100 % 100}{target % 100:x}"
def rocm_override(target: int | None) -> str | None:
"""The generation to present to ROCm, as HSA_OVERRIDE_GFX_VERSION, on a
GPU its images carry no kernels for but whose family they do: the RDNA2
ones (the Radeon 680M among them) as gfx1030 and the RDNA3 ones (the
Radeon 780M) as gfx1100. None when the GPU needs none or has none."""
if target is None or target in ROCM_TARGETS + ROCM_ACCELERATOR_TARGETS:
return None
if 100300 < target < 100400:
return "10.3.0"
if 110000 < target < 110100:
return "11.0.0"
return None
def rocm_support(target: int | None, targets=ROCM_TARGETS + ROCM_ACCELERATOR_TARGETS) -> str | None:
"""How an image with kernels for `targets` runs on this GPU: "native",
"experimental" when the GPU has to be presented as the generation of its
family, which ROCm does not support officially and can stop responding
under load, or None when it does not run. A driver that does not say the
generation does not rule the GPU out."""
if target is None or target in targets:
return "native"
override = rocm_override(target)
if override is None:
return None
major, minor, stepping = (int(part) for part in override.split("."))
return "experimental" if major * 10000 + minor * 100 + stepping in targets else None
def amd_gpu_name(root: Path = Path("/")) -> str:
"""The AMD GPU of the host as its owner knows it, with the generation ROCm
sees: "Radeon 680M (gfx1035)"."""
target = amd_gfx_target(root)
generation = gfx_name(target) if target else ""
name = ""
try:
listed = subprocess.run(["lspci", "-nn"], capture_output=True, text=True, timeout=5, check=False).stdout
except (OSError, subprocess.TimeoutExpired):
listed = ""
for line in listed.splitlines():
described = re.search(r"\[AMD/ATI\]\s*(.*?)\s*\[1002:", line)
if described and re.search(r"VGA|Display|3D", line):
marketed = re.search(r"\[([^\]]+)\]", described[1])
name = marketed[1] if marketed else described[1]
break
name = name or "AMD GPU"
return f"{name} ({generation})" if generation else name
def rocm_blocker(storage: str | None, needed_gb: int = 40) -> str | None:
"""Why this host cannot run recognition on an AMD GPU with ROCm, or None.
ROCm needs the compute interface of the driver and room for its image."""
ROCm needs the compute interface of the driver, a GPU of a generation its
image supports and room for that image."""
if not Path("/dev/kfd").is_char_device():
return "kfd"
if rocm_support(amd_gfx_target()) is None:
return "generation"
row = next((item for item in storages("rootdir") if item.get("storage") == storage), None)
if row is not None and gib(row.get("avail")) < needed_gb:
return "space"
+89 -34
View File
@@ -805,20 +805,45 @@ def _ask_immich_acceleration(ui, rootfs_storage: str | None = None) -> tuple[str
if not vendors:
real.message(translate("No usable GPU was found on this host. Immich will be installed on the CPU."))
return software
# Recognition on an AMD GPU depends on ROCm: it is offered when the host
# can run it, marked as experimental on a GPU ROCm does not support
# officially, and left out, with the reason, when it cannot.
amd_blocker = host.rocm_blocker(rootfs_storage) if found["amd"] else None
amd_experimental = (found["amd"] and not amd_blocker
and host.rocm_support(found.get("amd_gfx_target")) == "experimental")
if amd_blocker:
reason = (translate("The AMD driver does not offer its compute interface (/dev/kfd) on this host.")
if amd_blocker == "kfd" else
f"{translate('The ROCm image has no support for the AMD GPU of this host:')} {host.amd_gpu_name()}."
if amd_blocker == "generation" else
translate("The storage has less than 40 GB free for the ROCm image."))
real.message(f"{reason}\n\n{translate('Recognition is not offered on the AMD GPU; video transcoding is.')}")
mark = f" — {translate('experimental on this GPU')}" if amd_experimental else ""
options = [("cpu", translate("No acceleration (CPU)"))]
for vendor in vendors:
options += [(vendor, f"{names[vendor]}: {translate('video + recognition')}"),
(f"{vendor}-video", f"{names[vendor]}: {translate('video only')}"),
(f"{vendor}-ml", f"{names[vendor]}: {translate('recognition only')}")]
recognition = vendor != "amd" or not amd_blocker
suffix = mark if vendor == "amd" else ""
if recognition:
options.append((vendor, f"{names[vendor]}: {translate('video + recognition')}{suffix}"))
options.append((f"{vendor}-video", f"{names[vendor]}: {translate('video only')}"))
if recognition:
options.append((f"{vendor}-ml", f"{names[vendor]}: {translate('recognition only')}{suffix}"))
if found["nvidia"]:
options += [(f"{vendor}+nvidia", f"{names[vendor]}: {translate('video')} · NVIDIA: {translate('recognition')}")
for vendor in ("intel", "amd") if found[vendor]]
# The GPU matters for Immich, so the first one is proposed whole.
selected = real.choose(translate("Hardware acceleration for Immich"), options, vendors[0])
if selected is None:
raise UserCancelled(translate("Immich configuration cancelled"))
if selected == "cpu":
return software
# The GPU matters for Immich, so the first one is proposed whole; what is
# experimental or missing is never the proposal.
proposed = vendors[0]
if proposed == "amd" and (amd_blocker or amd_experimental):
proposed = "amd-video"
while True:
selected = real.choose(translate("Hardware acceleration for Immich"), options, proposed)
if selected is None:
raise UserCancelled(translate("Immich configuration cancelled"))
if selected == "cpu":
return software
if not (amd_experimental and selected in ("amd", "amd-ml")) or confirm_experimental_rocm(real):
break
if "+" in selected:
video_vendor, ml_vendor = selected.split("+", 1)
else:
@@ -848,20 +873,10 @@ def _ask_immich_acceleration(ui, rootfs_storage: str | None = None) -> tuple[str
elif ml_vendor == "intel":
ml_acceleration, ml_render = "openvino", render_device or found["intel"][0]
elif ml_vendor == "amd":
# ROCm is checked before it is promised; when the host cannot run it,
# recognition stays on the CPU.
blocker = host.rocm_blocker(rootfs_storage)
if blocker:
reason = (translate("The AMD driver does not offer its compute interface (/dev/kfd) on this host.")
if blocker == "kfd" else
translate("The storage has less than 40 GB free for the ROCm image."))
real.message(f"{reason}\n\n{translate('Recognition runs on the CPU.')}")
else:
ml_acceleration, ml_render = "rocm", render_device or found["amd"][0]
if ml_acceleration == "rocm":
ml_acceleration, ml_render = "rocm", render_device or found["amd"][0]
if ml_acceleration == "rocm" and not amd_experimental:
real.message(translate("Recognition on AMD uses ROCm. Its image is several times larger than the others, "
"so the first installation takes longer, and whether a GPU works with it depends "
"on its model."))
"so the first installation takes longer."))
if ml_acceleration != "cpu":
ui.message(translate("GPU recognition uses 8 GB of RAM and a limit of 4 CPU equivalents. These resources "
"were tested in the lab and are not a universal minimum. Compatibility depends on the "
@@ -962,6 +977,10 @@ def _build_immich_deployment(
"driver": vaapi_driver,
},
"machine_learning": {"acceleration": ml_acceleration, "render_device": ml_render,
# The generation ROCm is told to use on a GPU of a
# family its image supports, such as the 680M.
"gfx_override": (host.rocm_override(host.gpus().get("amd_gfx_target"))
if ml_acceleration == "rocm" else None),
"model_cache_size_gb": 8,
"resources": {"cores": 4,
"memory_mb": 4096 if ml_acceleration == "cpu" else 8192,
@@ -1428,7 +1447,7 @@ def _run_remote_install(
shutil.copy2(project_root / 'remote' / 'haos_healthcheck.py', temporary / 'haos_healthcheck.py')
shutil.copy2(project_root / 'remote' / 'oci_installation_state.py', temporary / 'oci_installation_state.py')
shutil.copy2(project_root / 'remote' / 'oci_instances.py', temporary / 'oci_instances.py')
for helper in ('oci_ui.sh', 'oci_ui.py', 'oci_description.py', 'oci_console.py', 'oci_native_stack.py', 'oci_native_stack.sh', 'oci_instance_transaction.py', 'oci_host_mounts.py', 'oci_runtime_settings.py', 'oci_gpu_devices.py', 'oci_accelerators.py', 'oci_nvidia_runtime.py', 'oci_nvidia_refresh.py', 'oci_nvidia_dynamic.py', 'oci_update_current.py', 'oci_stack_replay.py', 'oci_stack_plan.py', 'oci_stack_transaction.py', 'oci_stack_native.py', 'oci_image_cache.py', 'nvidia_lxc_mount_lab.sh', 'oci_nvidia_setup.sh', 'oci_immich_ml.sh'):
for helper in ('oci_ui.sh', 'oci_ui.py', 'oci_description.py', 'oci_console.py', 'oci_native_stack.py', 'oci_native_stack.sh', 'oci_instance_transaction.py', 'oci_host_mounts.py', 'oci_runtime_settings.py', 'oci_gpu_devices.py', 'oci_accelerators.py', 'oci_nvidia_runtime.py', 'oci_nvidia_refresh.py', 'oci_nvidia_dynamic.py', 'oci_update_current.py', 'oci_stack_replay.py', 'oci_stack_plan.py', 'oci_stack_transaction.py', 'oci_stack_native.py', 'oci_image_cache.py', 'nvidia_lxc_mount_lab.sh', 'oci_nvidia_setup.sh', 'oci_immich_ml.sh', 'oci_rocm_check.py'):
shutil.copy2(project_root / 'remote' / helper, temporary / helper)
if deployment_kind == 'generic-multi-lxc-stack':
shutil.copy2(project_root / 'remote' / 'install_generic_stack.py', temporary / 'install_generic_stack.py')
@@ -1465,7 +1484,7 @@ def _run_remote_install(
archive.add(project_root / 'remote' / 'haos_healthcheck.py', arcname='haos_healthcheck.py')
archive.add(project_root / 'remote' / 'oci_installation_state.py', arcname='oci_installation_state.py')
archive.add(project_root / 'remote' / 'oci_instances.py', arcname='oci_instances.py')
for helper in ('oci_ui.sh', 'oci_ui.py', 'oci_description.py', 'oci_console.py', 'oci_native_stack.py', 'oci_native_stack.sh', 'oci_instance_transaction.py', 'oci_host_mounts.py', 'oci_runtime_settings.py', 'oci_gpu_devices.py', 'oci_accelerators.py', 'oci_nvidia_runtime.py', 'oci_nvidia_refresh.py', 'oci_nvidia_dynamic.py', 'oci_update_current.py', 'oci_stack_replay.py', 'oci_stack_plan.py', 'oci_stack_transaction.py', 'oci_stack_native.py', 'oci_image_cache.py', 'nvidia_lxc_mount_lab.sh', 'oci_nvidia_setup.sh', 'oci_immich_ml.sh'):
for helper in ('oci_ui.sh', 'oci_ui.py', 'oci_description.py', 'oci_console.py', 'oci_native_stack.py', 'oci_native_stack.sh', 'oci_instance_transaction.py', 'oci_host_mounts.py', 'oci_runtime_settings.py', 'oci_gpu_devices.py', 'oci_accelerators.py', 'oci_nvidia_runtime.py', 'oci_nvidia_refresh.py', 'oci_nvidia_dynamic.py', 'oci_update_current.py', 'oci_stack_replay.py', 'oci_stack_plan.py', 'oci_stack_transaction.py', 'oci_stack_native.py', 'oci_image_cache.py', 'nvidia_lxc_mount_lab.sh', 'oci_nvidia_setup.sh', 'oci_immich_ml.sh', 'oci_rocm_check.py'):
archive.add(project_root / 'remote' / helper, arcname=helper)
if deployment_kind == 'generic-multi-lxc-stack':
archive.add(project_root / 'remote' / 'install_generic_stack.py', arcname='install_generic_stack.py')
@@ -1678,6 +1697,11 @@ def _profile_usable(profile: dict[str, Any], found: dict[str, Any]) -> bool:
"""Whether the host has what an acceleration profile needs: the NVIDIA
runtime, a GPU of the vendor it is written for, or ROCm's compute device."""
vendors = {"0x8086": "intel", "0x1002": "amd"}
# An image built for ROCm carries the kernels of some GPU generations
# only; on an older one its compute process aborts.
targets = profile.get("amd_gfx_targets")
if targets and host.rocm_support(found.get("amd_gfx_target"), tuple(targets)) is None:
return False
for request in profile.get("device_requests", []):
if request.get("kind") == "nvidia-runtime":
if not found["nvidia"]:
@@ -1694,6 +1718,20 @@ def _profile_usable(profile: dict[str, Any], found: dict[str, Any]) -> bool:
return True
def confirm_experimental_rocm(ui) -> bool:
"""What using ROCm means on a GPU it does not support officially, said
before the larger image is downloaded."""
return ui.confirm(
f"{host.amd_gpu_name()}\n\n"
+ translate("ROCm does not support this GPU officially. The application can use it by presenting it as "
"another generation of its family: it is faster than the CPU, but under load the GPU can stop "
"responding, and then the application does not answer until its container is restarted. The "
"ROCm image is also larger than the others.")
+ "\n\n" + translate("If it does not work well, the acceleration can be changed back from Manage installed "
"OCI applications, without reinstalling.")
+ "\n\n" + translate("Use the GPU with ROCm anyway?"), False)
def configure_acceleration(installer_profile, environment, unprivileged, ui, mode=ADVANCED_MODE):
advanced = mode != DEFAULT_MODE
devices: list[dict[str, Any]] = []
@@ -1710,22 +1748,35 @@ def configure_acceleration(installer_profile, environment, unprivileged, ui, mod
found = host.gpus()
usable = [item for item in profiles
if item["id"] == default_profile or _profile_usable(item, found)]
experimental = {item["id"] for item in usable if item.get("amd_gfx_targets") and host.rocm_support(
found.get("amd_gfx_target"), tuple(item["amd_gfx_targets"])) == "experimental"}
options = [(item["id"], item["label"]) for item in usable]
asked = advanced or not installer_profile.get("selkies")
if asked and found["amd"] and any(item.get("amd_gfx_targets") and item not in usable and host.rocm_support(
found.get("amd_gfx_target"), tuple(item["amd_gfx_targets"])) is None for item in profiles):
# Its owner may expect the GPU to be offered: say why it is not.
ui.message(f"{translate('The ROCm image of this application has no support for the AMD GPU of this host:')} "
f"{host.amd_gpu_name()}. {translate('Its ROCm profile is not offered.')}")
if asked and len(usable) < len(profiles) and len(usable) == 1:
# Nothing but the CPU is left: say why there is nothing to choose.
ui.message(translate("No usable GPU was found on this host. The application will be installed "
"without hardware acceleration."))
asked = False
selected_hardware_profile = (
ui.choose(
translate(hardware.get("prompt", "Hardware acceleration")),
[(tag, translate(label)) for tag, label in options],
default_profile,
) if asked else default_profile
)
if selected_hardware_profile is None:
raise UserCancelled(translate("Acceleration configuration cancelled"))
while True:
selected_hardware_profile = (
ui.choose(
translate(hardware.get("prompt", "Hardware acceleration")),
[(tag, translate(label) + (f" — {translate('experimental on this GPU')}" if tag in experimental else ""))
for tag, label in options],
default_profile,
) if asked else default_profile
)
if selected_hardware_profile is None:
raise UserCancelled(translate("Acceleration configuration cancelled"))
# The profile an application already uses is not asked about again.
if (selected_hardware_profile not in experimental or not asked
or selected_hardware_profile == default_profile or confirm_experimental_rocm(ui)):
break
selected_profile = next(
(item for item in profiles if item["id"] == selected_hardware_profile),
None,
@@ -1736,7 +1787,11 @@ def configure_acceleration(installer_profile, environment, unprivileged, ui, mod
{**item, "selected_by_hardware_profile": True}
for item in selected_profile.get("device_requests", [])
]
for env_item in selected_profile.get("environment", []):
profile_environment = list(selected_profile.get("environment", []))
if selected_hardware_profile in experimental:
profile_environment.append({"name": "HSA_OVERRIDE_GFX_VERSION",
"value": host.rocm_override(found.get("amd_gfx_target"))})
for env_item in profile_environment:
environment = [
entry for entry in environment if entry["name"] != env_item["name"]
]
+97
View File
@@ -125,11 +125,98 @@ def _run_lifecycle(command, title):
console.msg_title(title)
environment = dict(os.environ, OCI_SPINNER='1' if sys.stdout.isatty() else '0')
completed = subprocess.run(command, env=environment, check=False)
carry_records(images.PROJECT_ROOT)
if sys.stdin.isatty():
console.wait_for_enter(translate('Press Enter to return to the menu...'))
return completed.returncode == 0
def carry_records(project, mount_stopped=True, vmids=None, verify=False):
"""Leave inside each container the copy of its record that a restore on
another host needs. It is refreshed after every operation. Returns what
was found for each container."""
sys.path.insert(0, str(project / 'remote'))
try:
import oci_carried_record
return oci_carried_record.sync(oci_carried_record.instances.ROOT, vmids, mount_stopped, verify)
except (ImportError, OSError, ValueError, RuntimeError, subprocess.SubprocessError):
return {}
def recover_automatically(project):
"""Register, without asking, the containers whose record the cluster
keeps: the ones that migrated to this node. Nothing is started."""
if not restored_applications(project):
return
try:
subprocess.run([sys.executable, str(project / 'remote/oci_restore_recovery.py'), 'recover', '--automatic'],
capture_output=True, timeout=600, check=False)
except (OSError, subprocess.SubprocessError):
pass
def restored_applications(project):
"""Restored containers of this node that have no record on this host."""
sys.path.insert(0, str(project / 'remote'))
try:
import oci_restore_recovery
return oci_restore_recovery.pending(oci_restore_recovery.instances.ROOT)
except (ImportError, OSError, ValueError, RuntimeError):
return []
def _restored_firewall_rules(project):
"""The host firewall rules the restored host monitors had, as they would
be on this host."""
try:
result = subprocess.run([sys.executable, str(project / 'remote/oci_restore_recovery.py'), 'plan'],
capture_output=True, text=True, timeout=300, check=False)
plans = json.loads(result.stdout) if result.returncode == 0 else []
except (OSError, ValueError, subprocess.SubprocessError):
return []
return [rule for plan in plans if not plan.get('blockers') for rule in plan.get('firewall', [])]
def offer_recovery(project, ui):
"""A restored application cannot be managed until this host knows it
again: offer to register it before the list is shown."""
found = restored_applications(project)
if not found:
return
lines = '\n'.join(f" CT {row['vmid']} {row['hostname']}" for row in found)
text = (f"{translate('These containers were restored from a backup and this host has no record of their OCI application:')}"
f"\n\n{lines}\n\n"
f"{translate('Until they are registered again they cannot be updated or managed from here. The recovery checks first that everything they need is on this host and changes nothing otherwise.')}"
f"\n\n{translate('Recover them now?')}")
if not ui.confirm(text, default=True):
return
command = [sys.executable, str(project / 'remote/oci_restore_recovery.py'), 'recover']
for rule in _restored_firewall_rules(project):
if ui.confirm(translate('CT {vmid} is a host monitor and had a rule in the host firewall. Allow TCP port {port} from {subnet} through the firewall of this host? Existing firewall rules are not changed.').format(
vmid=rule['vmid'], port=rule['port'], subnet=rule['source']), default=False):
command.append('--host-firewall')
break
if ui.confirm(translate('Start the applications once they are registered? Answer No if the original containers are still running on another host: both would use the same addresses.'),
default=False):
command.append('--start')
_run_lifecycle(command, translate('Recover restored OCI applications'))
def direct_recovery(project):
"""The recovery of restored applications without the list of the menu: the
entry ProxMenux Monitor uses for its Recover button."""
from .ui import interactive_ui
ui = interactive_ui()
if os.geteuid() != 0 or not shutil.which('pct'):
ui.message(translate('This interface runs on the Proxmox node as root. Open OCI manager Apps from the ProxMenux menu on the Proxmox host.'), translate('OCI management'))
return EXIT_FAILED
if not restored_applications(project):
ui.message(translate('No restored OCI application is waiting to be recovered.'), translate('Restored OCI applications'))
return EXIT_DONE
offer_recovery(project, ui)
return EXIT_DONE if not restored_applications(project) else EXIT_FAILED
def interactive_management(project, ui):
try:
_interactive_management(project, ui)
@@ -157,6 +244,9 @@ def _interactive_management(project, ui):
ui.message(translate('This interface runs on the Proxmox node as root. Open OCI manager Apps from the ProxMenux menu on the Proxmox host.'), translate('OCI management'))
return
_clean_orphans(project)
recover_automatically(project)
offer_recovery(project, ui)
carry_records(project, mount_stopped=False)
rows = saved_inventory(project)
if not rows:
ui.message(translate('No registered OCI containers are available for selection on this host.'), translate('OCI management'))
@@ -184,6 +274,12 @@ def manage_instance(project, ui, row, action=None, lifecycle_args=()):
"""What the menu does with one instance once it is selected. `action`
skips the choice of operation, as ProxMenux Monitor does; the extra
`lifecycle_args` are passed to the program that performs it."""
# A container that came back from another node or from an older backup
# carries the record that describes it; the one of this host is not used.
if carry_records(project, vmids=[row['vmid']], verify=True).get(row['vmid']) == 'stale':
ui.message(translate('This container was restored or came back from another host after its record on this host was written. Open this menu again to recover it; nothing was changed.'),
translate('OCI management'))
return False
row = check_selected(project, row)
if row['reason'] != 'matched':
ui.message(translate('The selected CT does not match its OCI record. Its configuration will not be modified or deleted.'), translate('OCI management'))
@@ -500,6 +596,7 @@ def direct_management(project, vmid, action, lifecycle_args=(), unattended=False
ui.message(translate('Only the update of the image runs unattended.'), translate('OCI management'))
return EXIT_FAILED
try:
recover_automatically(project)
row = next((r for r in saved_inventory(project) if r['vmid'] == vmid), None)
if row is None:
ui.message(translate('This container is not a registered OCI instance.'), translate('OCI management'))
+71
View File
@@ -144,7 +144,78 @@ def summary(member, changes):
return '\n'.join(lines)
def immich_learning(primary):
"""The machine learning container of an Immich, or None for any other application."""
for member in primary.get('stack', {}).get('members', []):
if member.get('deployment', {}).get('replay_profile') == {'adapter': 'install_immich_stack.sh',
'role': 'machine-learning'}:
return member
return None
def recognition_choices(storage):
"""What can run the recognition of Immich on this host, each with the
devices it needs."""
from . import host
found = host.gpus()
choices = [('cpu', translate('CPU'), {})]
if found['intel']:
choices.append(('openvino', 'Intel GPU', {'render': found['intel'][0]}))
if found['nvidia']:
choices.append(('cuda', 'NVIDIA GPU', {}))
if found['amd'] and not host.rocm_blocker(storage):
experimental = host.rocm_support(found.get('amd_gfx_target')) == 'experimental'
label = 'AMD GPU' + (f" — {translate('experimental on this GPU')}" if experimental else '')
choices.append(('rocm', label, {'render': found['amd'][0], 'experimental': experimental,
'override': host.rocm_override(found.get('amd_gfx_target'))}))
return choices
def change_recognition(project, ui, primary, member, run_lifecycle):
"""Move the recognition of Immich between the CPU and a GPU of the host."""
from .installer import confirm_experimental_rocm
current = (member.get('deployment', {}).get('machine_learning') or {}).get('acceleration', 'cpu')
storage = _config(member['vmid']).get('rootfs', 'local-lvm:').split(':', 1)[0]
choices = recognition_choices(storage)
if [tag for tag, _, _ in choices] == ['cpu'] and current == 'cpu':
ui.message(translate('No usable GPU was found on this host. Recognition stays on the CPU.'),
translate('Recreate OCI'))
return False
selected = ui.choose(translate('What runs the recognition of Immich'),
[(tag, label) for tag, label, _ in choices], current)
if selected is None or selected == current:
ui.message(translate('Nothing was changed.'), translate('Recreate OCI'))
return False
details = next(extra for tag, _, extra in choices if tag == selected)
if details.get('experimental') and not confirm_experimental_rocm(ui):
return False
labels = {tag: label for tag, label, _ in choices}
text = (f"{translate('Recognition')}: {labels.get(current, current)} → {labels[selected]}\n\n"
+ translate('The machine learning container is rebuilt with the image of the new choice. Its model '
'cache, the library and the database are kept. The whole application is stopped and '
'updated, as in an update; if anything fails, the previous containers and the previous '
'choice are restored.'))
if not ui.review(text, translate('Recreate OCI'), question=translate('Change the recognition now?'), default=True):
return False
command = [sys.executable, str(project / 'remote/oci_immich_recognition.py'), str(primary['vmid']),
'--acceleration', selected]
if details.get('render'):
command += ['--render-device', details['render']]
if details.get('override'):
command += ['--gfx-override', details['override']]
return run_lifecycle(command, translate('Recreate OCI'))
def recreate_stack(project, ui, primary, run_lifecycle):
learning = immich_learning(primary)
if learning is not None:
what = ui.choose(translate('What to recreate'),
[('paths', translate('Add or remove extra paths and devices')),
('recognition', translate('Change what runs recognition: CPU or GPU'))], 'paths')
if what is None:
return False
if what == 'recognition':
return change_recognition(project, ui, primary, learning, run_lifecycle)
members = application_members(primary)
if not members:
ui.message(translate('This stack has no saved members to update.'), translate('OCI stack management'))
+81
View File
@@ -0,0 +1,81 @@
"""The record of an installation travels inside its container, so a backup
restored on another host still carries it."""
import json
import os
from pathlib import Path
import sys
import tempfile
import unittest
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
import oci_carried_record as carried
def bundle(**extra):
return {"schema_version": 1, "kind": carried.KIND, "node": "pve", "vmid": 120,
"record": {"vmid": 120, "installation_id": "488ed3cb-1145-477a-b90f-c46777d69fb2"}, **extra}
class CarriedRecordTests(unittest.TestCase):
def setUp(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
self.rootfs = Path(tmp.name) / "rootfs"
self.rootfs.mkdir()
self.outside = Path(tmp.name) / "outside"
self.outside.mkdir()
self.owner = (os.getuid(), os.getgid())
def test_the_copy_is_private_and_read_back_as_written(self):
carried.write_copy(self.rootfs, bundle(), self.owner)
path = self.rootfs / carried.DIRECTORY / carried.NAME
self.assertEqual(path.stat().st_mode & 0o777, 0o600)
self.assertEqual(path.parent.stat().st_mode & 0o777, 0o700)
self.assertEqual(carried.read_copy(self.rootfs), bundle())
self.assertEqual(os.listdir(path.parent), [carried.NAME])
def test_a_newer_record_replaces_the_copy(self):
carried.write_copy(self.rootfs, bundle(), self.owner)
carried.write_copy(self.rootfs, bundle(node="other"), self.owner)
self.assertEqual(carried.read_copy(self.rootfs)["node"], "other")
def test_a_link_left_in_place_of_the_folder_is_not_followed(self):
(self.rootfs / carried.DIRECTORY).symlink_to(self.outside)
with self.assertRaises(OSError):
carried.write_copy(self.rootfs, bundle(), self.owner)
self.assertEqual(os.listdir(self.outside), [])
(self.outside / carried.NAME).write_text(json.dumps(bundle()))
self.assertIsNone(carried.read_copy(self.rootfs))
def test_a_link_left_in_place_of_the_copy_is_not_followed(self):
(self.outside / "secret").write_text(json.dumps(bundle()))
(self.rootfs / carried.DIRECTORY).mkdir()
(self.rootfs / carried.DIRECTORY / carried.NAME).symlink_to(self.outside / "secret")
self.assertIsNone(carried.read_copy(self.rootfs))
def test_what_is_not_a_copy_is_not_read(self):
self.assertIsNone(carried.read_copy(self.rootfs))
folder = self.rootfs / carried.DIRECTORY
folder.mkdir()
for content in ("not json", json.dumps([1]), json.dumps(bundle(kind="other")),
json.dumps(bundle(schema_version=2)), json.dumps({**bundle(), "record": "text"})):
(folder / carried.NAME).write_text(content)
self.assertIsNone(carried.read_copy(self.rootfs), content)
def test_the_time_it_was_saved_does_not_make_a_copy_different(self):
self.assertEqual(carried.digest(bundle(saved_at="2026-10-03")), carried.digest(bundle(saved_at="2026-10-04")))
self.assertEqual(carried.digest(bundle(generation="a")), carried.digest(bundle(generation="b")))
self.assertNotEqual(carried.digest(bundle()), carried.digest(bundle(node="other")))
def test_the_owner_is_root_of_the_container(self):
self.assertEqual(carried.mapped_root("arch: amd64\nunprivileged: 1\n"), (100000, 100000))
self.assertEqual(carried.mapped_root("arch: amd64\n"), (0, 0))
custom = "unprivileged: 1\nlxc.idmap: u 0 200000 65536\nlxc.idmap: g 0 300000 65536\nlxc.idmap: u 1000 1000 1\n"
self.assertEqual(carried.mapped_root(custom), (200000, 300000))
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,47 @@
"""A container volume is always part of the backups: an update restores its
application from one, and refuses a volume that would be left out."""
import json
from pathlib import Path
import sys
import unittest
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
def volumes_without_backup(value, found=None):
"""Container paths a template would create as a container volume left out of the backups."""
found = [] if found is None else found
if isinstance(value, dict):
managed = value.get("managed_volume")
if (isinstance(managed, dict) and managed.get("backup") is not True
and "managed-volume" in (value.get("installation_choice") or [])):
found.append(value.get("container_path"))
if value.get("mode") == "managed-volume" and value.get("backup") is False:
found.append(value.get("container_path"))
for item in value.values():
volumes_without_backup(item, found)
elif isinstance(value, list):
for item in value:
volumes_without_backup(item, found)
return found
class ContainerVolumeBackupTests(unittest.TestCase):
def test_no_application_of_the_catalog_creates_a_volume_left_out_of_the_backups(self):
for folder in ("apps", "curated"):
for path in sorted((ROOT / "catalog" / folder).glob("*.json")):
template = json.loads(path.read_text(encoding="utf-8"))
self.assertEqual(volumes_without_backup(template), [], f"{folder}/{path.name}")
def test_a_converted_compose_backs_up_every_volume_it_creates(self):
from proxmenux_oci import converter
contract = converter._mount_contract(["/srv/app/config:/config", "/srv/app/cache:/cache",
"/srv/app/transcode:/transcode", "/srv/app/tmp:/tmp"], set())
self.assertEqual([item["container_path"] for item in contract], ["/config", "/cache", "/transcode", "/tmp"])
self.assertTrue(all(item["managed_volume"]["backup"] is True for item in contract))
if __name__ == "__main__":
unittest.main()
+143
View File
@@ -53,3 +53,146 @@ class FrigateAmdProfileTests(unittest.TestCase):
if __name__ == "__main__":
unittest.main()
class FrigateAmdGenerationTests(unittest.TestCase):
"""The ROCm image of Frigate carries the kernels of RDNA2 and newer GPUs:
on an older one, such as the integrated Vega of a Ryzen 5000U, its
detector aborts, so the profile is not offered there."""
def profile(self):
profiles = Catalog(ROOT).compose("frigate")["proxmox"]["installer_profile"]["hardware_acceleration"]["profiles"]
return next(item for item in profiles if item["id"] == "rocm")
def usable(self, target):
from proxmenux_oci.installer import _profile_usable
found = {"intel": [], "amd": ["/dev/dri/renderD128"], "nvidia": False, "amd_gfx_target": target}
with patch("proxmenux_oci.installer.Path.is_char_device", return_value=True):
return _profile_usable(self.profile(), found)
def test_the_profile_is_offered_on_the_generations_the_image_supports(self):
self.assertIn(100300, self.profile()["amd_gfx_targets"])
self.assertTrue(self.usable(100300))
self.assertTrue(self.usable(110501))
self.assertFalse(self.usable(90012))
self.assertFalse(self.usable(100100))
# The other GPUs of a family it supports are offered as experimental.
self.assertTrue(self.usable(100305))
self.assertTrue(self.usable(110003))
# A driver that does not say the generation does not hide the profile.
self.assertTrue(self.usable(None))
def test_the_generation_is_read_from_the_compute_driver(self):
import tempfile
from proxmenux_oci import host
with tempfile.TemporaryDirectory() as directory:
nodes = Path(directory) / "sys/class/kfd/kfd/topology/nodes"
for index, value in enumerate((0, 90012)):
(nodes / str(index)).mkdir(parents=True)
(nodes / str(index) / "properties").write_text(f"simd_count 32\ngfx_target_version {value}\ndevice_id 5708\n")
self.assertEqual(host.amd_gfx_target(Path(directory)), 90012)
self.assertIsNone(host.amd_gfx_target(Path(directory) / "none"))
self.assertEqual([host.gfx_name(value) for value in (90012, 100300, 110501, 90010, 90402)],
["gfx90c", "gfx1030", "gfx1151", "gfx90a", "gfx942"])
def test_recognition_with_rocm_needs_a_supported_generation(self):
from proxmenux_oci import host
with patch("proxmenux_oci.host.Path.is_char_device", return_value=True), \
patch("proxmenux_oci.host.storages", return_value=[]):
for target, expected in ((90012, "generation"), (100100, "generation"), (100305, None),
(110003, None), (110501, None), (90010, None), (None, None)):
with patch("proxmenux_oci.host.amd_gfx_target", return_value=target):
self.assertEqual(host.rocm_blocker("local-lvm"), expected, target)
def test_how_each_generation_runs_rocm(self):
from proxmenux_oci import host
expected = {100300: ("native", None), 110501: ("native", None), 90010: ("native", None),
100305: ("experimental", "10.3.0"), 100302: ("experimental", "10.3.0"),
110003: ("experimental", "11.0.0"), 90012: (None, None), 100100: (None, None),
110502: (None, None), None: ("native", None)}
for target, (support, override) in expected.items():
self.assertEqual((host.rocm_support(target), host.rocm_override(target)), (support, override), target)
# An image with fewer kernels supports fewer generations.
self.assertIsNone(host.rocm_support(110003, (100300,)))
def test_the_gpu_is_named_as_its_owner_knows_it(self):
from proxmenux_oci import host
from types import SimpleNamespace
listed = ("35:00.0 VGA compatible controller [0300]: Advanced Micro Devices, Inc. [AMD/ATI] "
"Rembrandt [Radeon 680M] [1002:1681] (rev c7)\n")
plain = "04:00.0 VGA compatible controller [0300]: Advanced Micro Devices, Inc. [AMD/ATI] Lucienne [1002:164c] (rev c1)\n"
for output, target, name in ((listed, 100305, "Radeon 680M (gfx1035)"), (plain, 90012, "Lucienne (gfx90c)"),
("", 100300, "AMD GPU (gfx1030)"), ("", None, "AMD GPU")):
with patch("proxmenux_oci.host.subprocess.run", return_value=SimpleNamespace(stdout=output)), \
patch("proxmenux_oci.host.amd_gfx_target", return_value=target):
self.assertEqual(host.amd_gpu_name(), name)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
@patch("proxmenux_oci.installer.Path.is_char_device", return_value=True)
@patch("proxmenux_oci.installer.host.amd_gpu_name", return_value="Radeon 680M (gfx1035)")
class FrigateExperimentalRocmTests(unittest.TestCase):
"""On a GPU ROCm does not support officially the profile is offered as
experimental, explained before the larger image is downloaded, and never
taken without a yes."""
WARNING = "ROCm does not support this GPU"
class UI(RecordingUI):
def __init__(self, answers, accept):
super().__init__(answers)
self.labels, self.messages, self.accept, self.asked_warning = {}, [], accept, 0
def choose(self, text, options, default=None):
self.labels[text] = dict(options)
return super().choose(text, options, default)
def confirm(self, text, default=False):
if FrigateExperimentalRocmTests.WARNING in text:
self.asked_warning += 1
return self.accept
return default
def message(self, text, title=None):
self.messages.append(text)
def build(self, target, answers, accept=True):
found = {"intel": [], "amd": ["/dev/dri/renderD128"], "nvidia": False, "amd_gfx_target": target}
ui = self.UI(answers, accept)
with patch("proxmenux_oci.installer.host.gpus", return_value=found):
plan = build_deployment(Catalog(ROOT).compose("frigate"), ui, DEFAULT_MODE)
return ui, plan
def override(self, plan):
return [item["value"] for item in plan["environment"] if item["name"] == "HSA_OVERRIDE_GFX_VERSION"]
def test_the_profile_is_marked_and_confirmed_and_sets_the_generation(self, *_):
ui, plan = self.build(100305, {PROMPT: "rocm"})
self.assertIn("experimental on this GPU", ui.labels[PROMPT]["rocm"])
self.assertEqual(ui.asked_warning, 1)
self.assertEqual(plan["hardware_profile"], "rocm")
self.assertEqual(self.override(plan), ["10.3.0"])
def test_a_supported_gpu_has_no_mark_and_no_question(self, *_):
ui, plan = self.build(100300, {PROMPT: "rocm"})
self.assertNotIn("experimental", ui.labels[PROMPT]["rocm"])
self.assertEqual((ui.asked_warning, self.override(plan)), (0, []))
def test_without_a_yes_the_menu_is_asked_again(self, *_):
answers = iter(["rocm", "vaapi"])
found = {"intel": [], "amd": ["/dev/dri/renderD128"], "nvidia": False, "amd_gfx_target": 100305}
ui = self.UI({}, False)
ui.choose = lambda text, options, default=None: next(answers) if text == PROMPT else default
with patch("proxmenux_oci.installer.host.gpus", return_value=found):
plan = build_deployment(Catalog(ROOT).compose("frigate"), ui, DEFAULT_MODE)
self.assertEqual(plan["hardware_profile"], "vaapi")
self.assertEqual(self.override(plan), [])
def test_a_gpu_the_image_cannot_use_is_not_offered_and_the_reason_is_said(self, *_):
with patch("proxmenux_oci.installer.host.amd_gpu_name", return_value="Lucienne (gfx90c)"):
ui, plan = self.build(90012, {})
self.assertNotIn("rocm", ui.labels[PROMPT])
self.assertTrue(any("Lucienne (gfx90c)" in message and "not offered" in message for message in ui.messages))
+48 -5
View File
@@ -91,14 +91,57 @@ class ImmichAccelerationTests(unittest.TestCase):
self.assertIn("No usable GPU", ui.messages[0])
self.assertEqual(result, ("cpu", "cpu"), mode)
def test_an_amd_host_that_cannot_run_rocm_says_so_and_recognises_on_the_cpu(self, *_):
for blocker, text in (("kfd", "/dev/kfd"), ("space", "40 GB")):
with patch("proxmenux_oci.installer.host.rocm_blocker", return_value=blocker):
ui, result, _ = self.plan(AMD, DEFAULT_MODE, "amd")
def test_an_amd_host_that_cannot_run_rocm_offers_video_only_and_says_why(self, *_):
for blocker, text in (("kfd", "/dev/kfd"), ("space", "40 GB"), ("generation", "no support for the AMD GPU")):
with patch("proxmenux_oci.installer.host.rocm_blocker", return_value=blocker), \
patch("proxmenux_oci.installer.host.amd_gpu_name", return_value="Lucienne (gfx90c)"):
ui, result, _ = self.plan(AMD, DEFAULT_MODE)
self.assertEqual(ui.options[PROMPT], ["cpu", "amd-video"], blocker)
self.assertEqual(ui.defaults[PROMPT], "amd-video", blocker)
self.assertEqual(result, ("vaapi", "cpu"), blocker)
self.assertTrue(any(text in message and "Recognition runs on the CPU." in message
self.assertTrue(any(text in message and "Recognition is not offered" in message
for message in ui.messages), blocker)
def test_a_gpu_rocm_supports_is_proposed_whole(self, *_):
for target in (100300, 110501, None):
ui, result, plan = self.plan(dict(AMD, amd_gfx_target=target), DEFAULT_MODE)
self.assertEqual(ui.defaults[PROMPT], "amd", target)
self.assertEqual(result, ("vaapi", "rocm"), target)
self.assertIsNone(plan["machine_learning"]["gfx_override"], target)
def test_a_gpu_rocm_does_not_support_officially_is_offered_as_experimental(self, *_):
gpus = dict(AMD, amd_gfx_target=100305)
with patch("proxmenux_oci.installer.host.amd_gpu_name", return_value="Radeon 680M (gfx1035)"):
# Never the proposal: left alone, recognition stays on the CPU.
ui, result, _ = self.plan(gpus, DEFAULT_MODE)
self.assertEqual(ui.options[PROMPT], ["cpu", "amd", "amd-video", "amd-ml"])
self.assertEqual(ui.defaults[PROMPT], "amd-video")
self.assertEqual(result, ("vaapi", "cpu"))
# Chosen and confirmed, it is installed with the generation of its family.
ui = OptionsUI({PROMPT: "amd"})
ui.confirm = lambda text, default=False: True if "ROCm does not support this GPU" in text else default
with patch("proxmenux_oci.installer.host.gpus", return_value=gpus):
plan = build_deployment(self.template, ui, DEFAULT_MODE)
self.assertEqual(plan["machine_learning"]["acceleration"], "rocm")
self.assertEqual(plan["machine_learning"]["gfx_override"], "10.3.0")
# Declined, the menu is asked again.
answers = iter(["amd", "amd-video"])
ui = OptionsUI()
ui.choose = lambda text, options, default=None: next(answers) if text == PROMPT else default
with patch("proxmenux_oci.installer.host.gpus", return_value=gpus):
plan = build_deployment(self.template, ui, DEFAULT_MODE)
self.assertEqual(plan["machine_learning"]["acceleration"], "cpu")
self.assertEqual(plan["video_transcoding"]["acceleration"], "vaapi")
def test_the_780m_family_is_presented_as_its_generation(self, *_):
gpus = dict(AMD, amd_gfx_target=110003)
ui = OptionsUI({PROMPT: "amd-ml"})
ui.confirm = lambda text, default=False: True if "ROCm does not support this GPU" in text else default
with patch("proxmenux_oci.installer.host.gpus", return_value=gpus), \
patch("proxmenux_oci.installer.host.amd_gpu_name", return_value="Radeon 780M (gfx1103)"):
plan = build_deployment(self.template, ui, DEFAULT_MODE)
self.assertEqual(plan["machine_learning"]["gfx_override"], "11.0.0")
def test_machine_learning_gets_four_cores_and_at_least_four_gigabytes(self, *_):
_, _, plan = self.plan(BOTH, DEFAULT_MODE, "cpu")
self.assertEqual(plan["machine_learning"]["resources"]["cores"], 4)
+288
View File
@@ -0,0 +1,288 @@
"""The recognition of an installed Immich moves between the CPU and a GPU of
the host without reinstalling: the machine learning container takes the image
and the devices of the new choice."""
from pathlib import Path
import sys
import tempfile
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
sys.path.insert(0, str(ROOT / "src"))
import oci_immich_recognition as recognition
import oci_rocm_check
CONFIG = """arch: amd64
cores: 4
dev0: path=/dev/dri/renderD128,gid=993,mode=0660
dev1: path=/dev/kfd,gid=993,mode=0660
memory: 8192
lxc.environment.runtime: IMMICH_PORT=3003
lxc.environment.runtime: HSA_OVERRIDE_GFX_VERSION=10.3.0
lxc.environment.runtime: HSA_USE_SVM=0
lxc.environment: NVIDIA_VISIBLE_DEVICES=all
lxc.hook.mount: /usr/local/lib/proxmenux/oci/nvidia-mount-abc.sh
lxc.signal.halt: SIGTERM
[snapshot]
arch: amd64
lxc.environment.runtime: HSA_USE_SVM=0
"""
class ConfigurationTests(unittest.TestCase):
def setUp(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
self.path = Path(tmp.name) / "101.conf"
self.path.write_text(CONFIG)
patcher = patch.object(recognition, "conf", return_value=self.path)
patcher.start()
self.addCleanup(patcher.stop)
def test_what_the_previous_choice_left_is_taken_away(self):
deleted = []
with patch.object(recognition.oci_stack_modify, "entries",
return_value={"dev0": "path=/dev/dri/renderD128,gid=993", "dev1": "path=/dev/kfd,gid=993",
"dev2": "path=/dev/ttyUSB0,gid=20"}), \
patch.object(recognition, "run", side_effect=lambda *command: deleted.append(command[-1])):
recognition.strip_gpu(101)
self.assertEqual(deleted, ["dev0", "dev1"])
text = self.path.read_text()
current, _, snapshot = text.partition("\n[")
for gone in ("HSA_OVERRIDE", "HSA_USE_SVM", "NVIDIA_VISIBLE_DEVICES", "nvidia-mount"):
self.assertNotIn(gone, current)
self.assertIn("lxc.environment.runtime: IMMICH_PORT=3003\n", current)
self.assertIn("lxc.signal.halt: SIGTERM\n", current)
# A snapshot is history: it is not rewritten.
self.assertEqual(snapshot, CONFIG.partition("\n[")[2])
def test_new_settings_go_before_the_snapshots(self):
recognition.append_lines(101, ["lxc.environment.runtime: HSA_OVERRIDE_GFX_VERSION=11.0.0"])
current, _, snapshot = self.path.read_text().partition("\n[")
self.assertTrue(current.endswith("lxc.environment.runtime: HSA_OVERRIDE_GFX_VERSION=11.0.0\n"))
self.assertEqual(snapshot, CONFIG.partition("\n[")[2])
class RecordTests(unittest.TestCase):
RECORD = {"deployment": {"machine_learning": {
"acceleration": "openvino", "render_device": "/dev/dri/renderD128", "model_cache_size_gb": 8,
"resources": {"cores": 4, "memory_mb": 8192, "swap_mb": 1024, "cpu_allocation": "quota"}}}}
def test_the_record_says_what_runs_recognition_and_with_what_resources(self):
cpu = recognition.recorded(self.RECORD, "cpu", None, None)
self.assertEqual((cpu["acceleration"], cpu["render_device"], cpu["gfx_override"]), ("cpu", None, None))
self.assertEqual((cpu["resources"]["memory_mb"], cpu["resources"]["cpu_allocation"]), (4096, "cpuset"))
self.assertEqual((cpu["model_cache_size_gb"], cpu["resources"]["swap_mb"]), (8, 1024))
rocm = recognition.recorded(self.RECORD, "rocm", "/dev/dri/renderD129", "10.3.0")
self.assertEqual((rocm["acceleration"], rocm["render_device"], rocm["gfx_override"]),
("rocm", "/dev/dri/renderD129", "10.3.0"))
self.assertEqual(rocm["resources"]["memory_mb"], 8192)
self.assertEqual(recognition.recorded(self.RECORD, "openvino", "/dev/dri/renderD128", None)["resources"]
["cpu_allocation"], "quota")
# The record it was read from is not changed.
self.assertEqual(self.RECORD["deployment"]["machine_learning"]["acceleration"], "openvino")
def test_the_record_declares_the_variables_of_the_new_choice(self):
base = [{"name": "IMMICH_PORT", "value": "3003"}]
names = lambda found: [(entry["name"], entry["value"]) for entry in found]
nvidia = recognition.declared(base, "cuda", None)
self.assertEqual(names(nvidia), [("IMMICH_PORT", "3003"), ("NVIDIA_DRIVER_CAPABILITIES", "compute,utility")])
amd = recognition.declared(nvidia, "rocm", "10.3.0")
self.assertEqual(names(amd), [("IMMICH_PORT", "3003"), ("HSA_OVERRIDE_GFX_VERSION", "10.3.0"),
("HSA_USE_SVM", "0")])
self.assertEqual(names(recognition.declared(nvidia, "rocm", None)), names(base))
self.assertEqual(names(recognition.declared(amd, "cpu", None)), names(base))
self.assertEqual(names(recognition.declared(amd, "openvino", None)), names(base))
def test_the_record_declares_the_resources_of_the_new_choice(self):
intel = {"cores": 4, "memory_mb": 8192, "swap_mb": 1024, "cpu_allocation": "quota"}
self.assertEqual(recognition.resources(intel, "cpu"), {"cores": 4, "memory_mb": 4096, "swap_mb": 1024})
self.assertEqual(recognition.resources(intel, "cuda"), {"cores": 4, "memory_mb": 8192, "swap_mb": 1024})
self.assertEqual(recognition.resources(recognition.resources(intel, "cpu"), "openvino"), intel)
self.assertEqual(intel["cpu_allocation"], "quota")
def test_a_record_rebuilt_by_an_update_describes_the_new_container(self):
def record(**plan):
return {"vmid": 101, "template": {"proxmox": {"installer_profile": {"id": "ml", "cpu_allocation": "quota"}}},
"deployment": {"environment": [], "rootfs": {"storage": "local-lvm", "size_gb": 12},
"resources": {"cores": 4, "memory_mb": 8192, "cpu_allocation": "quota"}, **plan}}
with patch.object(recognition, "settings", return_value={"rootfs": "local-lvm:vm-101-disk-0,size=40G"}):
rebuilt = record()
recognition.declare(rebuilt, "rocm", "11.0.0")
self.assertEqual(rebuilt["deployment"]["rootfs"]["size_gb"], 40)
self.assertEqual(rebuilt["deployment"]["resources"], {"cores": 4, "memory_mb": 8192})
self.assertEqual(rebuilt["template"]["proxmox"]["installer_profile"], {"id": "ml"})
self.assertEqual([entry["name"] for entry in rebuilt["deployment"]["environment"]],
["HSA_OVERRIDE_GFX_VERSION", "HSA_USE_SVM"])
recognition.declare(rebuilt, "openvino", None)
self.assertEqual(rebuilt["template"]["proxmox"]["installer_profile"]["cpu_allocation"], "quota")
self.assertEqual(rebuilt["deployment"]["environment"], [])
# A record still described by its native configuration is read from it.
native = record(native_config="arch: amd64")
recognition.declare(native, "rocm", "11.0.0")
self.assertEqual(native, record(native_config="arch: amd64"))
def test_a_choice_this_host_cannot_serve_is_refused_before_anything_changes(self):
with self.assertRaises(ValueError):
recognition.validate("vulkan", None, None)
with self.assertRaises(ValueError):
recognition.validate("rocm", "/dev/dri/renderD999", None)
with self.assertRaises(ValueError):
recognition.validate("openvino", "/etc/passwd", None)
with self.assertRaises(ValueError):
recognition.validate("cpu", None, "10.3.0")
recognition.validate("cpu", None, None)
def test_only_an_immich_of_proxmenux_is_changed(self):
def member(vmid, adapter, role):
return {"vmid": vmid, "deployment": {"replay_profile": {"adapter": adapter, "role": role}}}
def records(roles, adapter="install_immich_stack.sh"):
found = {vmid: member(vmid, adapter, role) for vmid, role in roles.items()}
found[100]["stack"] = {"members": [{"vmid": vmid} for vmid in roles]}
return found
immich = records({100: "server", 101: "machine-learning", 102: "database", 103: "valkey"})
with patch.object(recognition.instances, "read", side_effect=lambda root, vmid: immich[vmid]):
_, found = recognition.members(Path("/nonexistent"), 100)
self.assertEqual(found["machine-learning"]["vmid"], 101)
for other in (records({100: "application", 101: "database"}, "install_tandoor_stack.sh"),
records({100: "server", 101: "machine-learning"})):
with patch.object(recognition.instances, "read", side_effect=lambda root, vmid: other[vmid]):
with self.assertRaises(ValueError):
recognition.members(Path("/nonexistent"), 100)
class RevertTests(unittest.TestCase):
def test_the_previous_choice_is_what_the_record_says(self):
record = {"deployment": {"machine_learning": {"acceleration": "rocm", "render_device": "/dev/dri/renderD128",
"gfx_override": "10.3.0"}}}
self.assertEqual(recognition.chosen(record), ("rocm", "/dev/dri/renderD128", "10.3.0"))
self.assertEqual(recognition.chosen({"deployment": {}}), ("cpu", None, None))
def test_a_failed_change_puts_the_previous_choice_back_the_way_it_was_given(self):
calls = []
with patch.object(recognition.oci_stack_modify, "is_running", return_value=True), \
patch.object(recognition, "stop", side_effect=lambda vmid: calls.append(("stop", vmid))), \
patch.object(recognition, "apply", side_effect=lambda *arguments: calls.append(("apply",) + arguments[2:])), \
patch.object(recognition.subprocess, "run", side_effect=lambda command, **_: calls.append(tuple(command[:3]))):
recognition.revert(Path("/nonexistent"), 100, 101, {"cpu": "image"}, ("cpu", None, None), True)
self.assertEqual(calls, [("stop", 101), ("apply", 101, {"cpu": "image"}, "cpu", None, None),
("pct", "start", "101")])
def test_a_revert_that_fails_does_not_hide_the_first_error(self):
with patch.object(recognition.oci_stack_modify, "is_running", return_value=False), \
patch.object(recognition, "apply", side_effect=RuntimeError("pct set: locked")), \
patch.object(recognition, "msg_warn") as warned, \
patch.object(recognition.subprocess, "run") as started:
recognition.revert(Path("/nonexistent"), 100, 101, {}, ("cpu", None, None), True)
self.assertIn("pct set: locked", warned.call_args[0][0])
started.assert_not_called()
class RocmProbeTests(unittest.TestCase):
def test_the_probe_runs_with_what_the_container_tells_rocm(self):
config = ("arch: amd64\nlxc.environment.runtime: PATH=/opt/venv/bin\n"
"lxc.environment.runtime: HSA_OVERRIDE_GFX_VERSION=10.3.0\nlxc.environment.runtime: HSA_USE_SVM=0\n"
"lxc.environment.runtime: HSA_BAD=1; reboot\n")
self.assertEqual(oci_rocm_check.variables(config), ["HSA_OVERRIDE_GFX_VERSION=10.3.0", "HSA_USE_SVM=0"])
self.assertEqual(oci_rocm_check.variables("arch: amd64\n"), [])
listed = "arch: amd64\nenv: PATH=/opt/venv/bin\0HSA_OVERRIDE_GFX_VERSION=11.0.0\0HSA_USE_SVM=0\0HOME=/root\n"
self.assertEqual(oci_rocm_check.variables(listed), ["HSA_OVERRIDE_GFX_VERSION=11.0.0", "HSA_USE_SVM=0"])
def protobuf_fields(data):
"""The fields of a protobuf message as (number, value) pairs."""
position = 0
def varint():
nonlocal position
value = shift = 0
while True:
byte = data[position]
position += 1
value |= (byte & 0x7F) << shift
shift += 7
if not byte & 0x80:
return value
while position < len(data):
key = varint()
number, wire = key >> 3, key & 7
if wire == 0:
yield number, varint()
elif wire == 2:
size = varint()
yield number, data[position:position + size]
position += size
else:
position += 4 if wire == 5 else 8
def packed_numbers(data):
numbers, value, shift = [], 0, 0
for byte in data:
value |= (byte & 0x7F) << shift
shift += 7
if not byte & 0x80:
numbers.append(value)
value = shift = 0
return numbers
class RocmProbeModelTests(unittest.TestCase):
def test_the_embedded_model_has_layers_that_fit_each_other(self):
import base64
model = base64.b64decode(oci_rocm_check.MODEL, validate=True)
graph = next(value for number, value in protobuf_fields(model) if number == 7)
shapes = {}
for number, tensor in protobuf_fields(graph):
if number != 5:
continue
dims, name = [], None
for field, value in protobuf_fields(tensor):
if field == 1:
dims += packed_numbers(value) if isinstance(value, bytes) else [value]
elif field == 8:
name = value.decode()
shapes[name] = dims
# The dense layer takes one value per channel of the convolution.
self.assertEqual(shapes["w"], [4, 3, 3, 3])
self.assertEqual(shapes["fw"], [shapes["w"][0], 2])
self.assertEqual((shapes["b"], shapes["fb"]), ([4], [2]))
class ChoicesTests(unittest.TestCase):
def choices(self, found, blocker=None):
from proxmenux_oci import stack_recreation
with patch("proxmenux_oci.host.gpus", return_value=found), \
patch("proxmenux_oci.host.rocm_blocker", return_value=blocker):
return stack_recreation.recognition_choices("local-lvm")
def test_the_choices_are_what_the_host_has(self):
node = "/dev/dri/renderD128"
tags = lambda found, blocker=None: [tag for tag, _, _ in self.choices(found, blocker)]
self.assertEqual(tags({"intel": [], "amd": [], "nvidia": False}), ["cpu"])
self.assertEqual(tags({"intel": [node], "amd": [], "nvidia": True}), ["cpu", "openvino", "cuda"])
self.assertEqual(tags({"intel": [], "amd": [node], "nvidia": False, "amd_gfx_target": 100300}), ["cpu", "rocm"])
self.assertEqual(tags({"intel": [], "amd": [node], "nvidia": False, "amd_gfx_target": 90012}, "generation"), ["cpu"])
def test_an_amd_gpu_rocm_does_not_support_officially_is_marked(self):
found = {"intel": [], "amd": ["/dev/dri/renderD128"], "nvidia": False, "amd_gfx_target": 100305}
tag, label, details = self.choices(found)[-1]
self.assertEqual(tag, "rocm")
self.assertIn("experimental", label)
self.assertEqual((details["experimental"], details["override"], details["render"]),
(True, "10.3.0", "/dev/dri/renderD128"))
native = self.choices(dict(found, amd_gfx_target=110501))[-1]
self.assertEqual((native[2]["experimental"], native[2]["override"]), (False, None))
self.assertNotIn("experimental", native[1])
if __name__ == "__main__":
unittest.main()
+9
View File
@@ -32,6 +32,8 @@ class RemovalCleanupTests(unittest.TestCase):
self.snippets.mkdir()
self.apps = self.root / "apps"
self.apps.mkdir()
self.records = self.root / "cluster-records"
self.records.mkdir()
self.cluster = self.root / "proxmenux"
self.cluster.mkdir()
self.legacy = self.root / "legacy"
@@ -41,6 +43,7 @@ class RemovalCleanupTests(unittest.TestCase):
patch.object(oci_remove, "CLUSTER_NODES", self.nodes),
patch.object(oci_remove, "SNIPPETS", self.snippets),
patch.object(oci_remove, "MONITOR_APPS", self.apps),
patch.object(oci_remove, "CLUSTER_RECORDS", self.records),
patch.object(oci_remove, "HOST_MONITOR_INCLUDES", self.host_monitor),
patch.object(runtime_settings, "include_path", lambda vmid: self.cluster / f"{vmid}.sysctls"),
patch.object(runtime_settings, "legacy_include_path", lambda vmid: self.legacy / f"{vmid}.proxmenux-sysctls"),
@@ -67,6 +70,12 @@ class RemovalCleanupTests(unittest.TestCase):
self.assertFalse((self.apps / "113.json").exists())
self.assertEqual(json.loads((self.apps / ".oci-dismissed.json").read_text()), {"112": "b"})
def test_the_copy_of_the_record_in_the_cluster_goes_with_the_container(self):
(self.records / "120.json").write_text("{}")
(self.records / "121.json").write_text("{}")
oci_remove.remove_host_state(120)
self.assertEqual([path.name for path in self.records.iterdir()], ["121.json"])
def test_a_published_view_with_content_is_kept(self):
shared = self.root / "shared"
(shared / "rw/drive").mkdir(parents=True)
+550
View File
@@ -0,0 +1,550 @@
"""A container restored from a backup is found, checked against what this
host has, and registered again only when its whole application can be."""
import copy
import ipaddress
import os
from pathlib import Path
import sys
import tempfile
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
import oci_carried_record as carried
import oci_instances as instances
import oci_restore_recovery as recovery
APP = "83e373f2-5ccc-4548-9b89-22d0956d1c77"
DB = "488ed3cb-1145-477a-b90f-c46777d69fb2"
def saved(identity, hostname, extra=""):
return f"#<div>Tandoor</div>\n#<!-- proxmenux-instance={identity} -->\narch: amd64\nhostname: {hostname}\n{extra}"
def config(identity, extra=""):
return (f"arch: amd64\ndescription: <!-- proxmenux-instance={identity} -->\nhostname: app\nmemory: 2048\n"
f"rootfs: local-lvm:vm-119-disk-0,size=8G\nunprivileged: 1\n{extra}")
CONSOLE = ("lxc.console.logfile: /var/log/proxmenux/oci/{0}.console.log\n"
"lxc.hook.pre-start: /bin/sh -c 'mkdir -p /var/log/proxmenux/oci; "
"test -x /usr/local/share/proxmenux/oci/engine/remote/oci_console_mark.sh "
"&& /usr/local/share/proxmenux/oci/engine/remote/oci_console_mark.sh {0}; exit 0'\n")
def record(vmid, identity, **extra):
return {"schema_version": 1, "vmid": vmid, "installation_id": identity, "status": "installed",
"template": {"catalog_ui": {"title": {"en_US": "Tandoor Recipes"}}},
"deployment": {"vmid": vmid, "hostname": "app", "rootfs": {"storage": "local-lvm", "size_gb": 8}},
"observed": {"config": config(identity), "config_sha256": "old", "api_config_sha256": "old",
"image": {"manifest_digest": "sha256:1"}, "archive_path": "/gone.tar",
"resolved_registry_digest": "sha256:1"}, **extra}
def entry(vmid, identity, value, text=None):
return {"vmid": vmid, "installation_id": identity, "hostname": f"ct{vmid}", "problem": None,
"config": text if text is not None else config(identity),
"copy": {"schema_version": 1, "kind": carried.KIND, "node": "old", "vmid": vmid, "record": value}}
class DiscoveryTests(unittest.TestCase):
def setUp(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
self.root = Path(tmp.name)
def register(self, vmid, identity):
with patch.object(instances, "private_directory", lambda path: path.mkdir(parents=True, exist_ok=True)):
instances.write(instances.location(self.root, vmid), record(vmid, identity))
def pending(self, guests):
with patch.object(recovery, "local_guests", return_value=guests):
return [row["vmid"] for row in recovery.pending(self.root)]
def test_the_mark_is_read_from_the_saved_configuration(self):
self.assertEqual(recovery.marker(saved(APP, "app")), APP)
self.assertIsNone(recovery.marker("arch: amd64\nhostname: plain\n"))
def test_a_marked_container_without_record_waits_for_recovery(self):
guests = {119: saved(APP, "app"), 120: saved(DB, "db"), 100: "arch: amd64\nhostname: plain\n"}
self.assertEqual(self.pending(guests), [119, 120])
self.register(119, APP)
self.assertEqual(self.pending(guests), [120])
def test_the_record_of_another_installation_does_not_count(self):
self.register(119, DB)
self.assertEqual(self.pending({119: saved(APP, "app")}), [119])
def test_a_container_that_carries_no_record_is_not_offered_again(self):
guests = {119: saved(APP, "app"), 120: saved(DB, "db")}
with patch.object(instances, "private_directory", lambda path: path.mkdir(parents=True, exist_ok=True)), \
patch.object(recovery, "local_guests", return_value=guests):
recovery.set_unrecoverable(self.root, [{"vmid": 119, "installation_id": APP}])
self.assertEqual(self.pending(guests), [120])
# Another installation restored on that ID is a new case.
self.assertEqual(self.pending({119: saved(DB, "other")}), [119])
def test_a_clone_of_a_registered_container_is_left_alone(self):
self.register(119, APP)
self.assertEqual(self.pending({119: saved(APP, "app"), 130: saved(APP, "copy")}), [])
class ApplicationTests(unittest.TestCase):
def stack(self):
database = record(120, DB, stack_member={"stack_id": APP, "primary_vmid": 119, "name": "database"})
application = record(119, APP, stack_member={"stack_id": APP, "primary_vmid": 119, "name": "application"})
application["stack"] = {"id": APP, "template": {"catalog_ui": {"title": {"en_US": "Tandoor Recipes"}}},
"members": [copy.deepcopy(application), copy.deepcopy(database)]}
return application, database
def plans(self, found, guests, known=None):
with patch.object(recovery, "local_guests", return_value=guests), \
patch.object(recovery, "registered_identities", return_value=known or {}):
return recovery.applications(Path("/nonexistent"), found)
def test_the_containers_of_one_application_are_recovered_together(self):
application, database = self.stack()
plans = self.plans([entry(120, DB, database), entry(119, APP, application)],
{119: saved(APP, "app"), 120: saved(DB, "db")})
self.assertEqual(len(plans), 1)
self.assertEqual((plans[0]["title"], plans[0]["primary"], plans[0]["blockers"]), ("Tandoor Recipes", 119, []))
self.assertEqual([m["state"] for m in plans[0]["members"]], ["restored", "restored"])
def test_an_application_with_a_container_missing_is_not_recovered(self):
application, _ = self.stack()
plans = self.plans([entry(119, APP, application)], {119: saved(APP, "app")})
self.assertEqual(len(plans[0]["blockers"]), 1)
self.assertIn("CT 120 (database)", plans[0]["blockers"][0])
def test_containers_restored_with_other_ids_keep_their_application(self):
application, database = self.stack()
application["stack"]["deployment"] = {"base_vmid": 119, "services": [
{"vmid": 119, "name": "application"},
{"vmid": 120, "name": "database", "healthcheck": {"timeout_seconds": 120}}]}
application["deployment"]["create_arguments"] = ["119", "local:vztmpl/app.tar"]
application["observed"]["config"] = config(APP, CONSOLE.format(119))
found = [entry(140, APP, application), entry(135, DB, database)]
found[0]["copy"]["stack_contract"] = {"schema": 1, "dependencies": [
{"vmid": 120, "label": "PostgreSQL", "healthcheck": {"type": "running", "timeout_seconds": 120}}]}
plans = self.plans(found, {140: saved(APP, "app"), 135: saved(DB, "db")})
plan = plans[0]
self.assertEqual((plan["blockers"], plan["renumbered"], plan["primary"]), ([], {119: 140, 120: 135}, 140))
self.assertEqual([m["vmid"] for m in plan["members"]], [140, 135])
main, other = (e["copy"]["record"] for e in plan["restored"])
self.assertEqual((main["vmid"], main["deployment"]["vmid"], main["stack_member"]["primary_vmid"]), (140, 140, 140))
self.assertEqual((other["vmid"], other["stack_member"]["primary_vmid"]), (135, 140))
self.assertEqual([m["vmid"] for m in main["stack"]["members"]], [140, 135])
self.assertEqual(main["stack"]["deployment"]["base_vmid"], 140)
services = main["stack"]["deployment"]["services"]
self.assertEqual([s["vmid"] for s in services], [140, 135])
# A number that is not an ID stays: 120 seconds is not CT 120.
self.assertEqual(services[1]["healthcheck"]["timeout_seconds"], 120)
contract = plan["restored"][0]["copy"]["stack_contract"]["dependencies"][0]
self.assertEqual((contract["vmid"], contract["healthcheck"]["timeout_seconds"]), (135, 120))
self.assertEqual(main["deployment"]["create_arguments"][0], "140")
self.assertIn("/var/log/proxmenux/oci/140.console.log", main["observed"]["config"])
self.assertIn("oci_console_mark.sh 140; exit 0", main["observed"]["config"])
self.assertNotIn("119", main["observed"]["config"].replace("vm-119-disk", ""))
self.assertEqual(plan["restored"][0]["recorded"], 119)
def test_the_lines_of_a_configuration_that_carry_the_id(self):
text = config(APP, CONSOLE.format(119) + "lxc.include: /etc/pve/proxmenux/119.sysctls\nmemory: 119\n")
result = recovery.renumbered_lines(text, 119, 140)
self.assertIn("lxc.console.logfile: /var/log/proxmenux/oci/140.console.log\n", result)
self.assertIn("oci_console_mark.sh 140; exit 0'\n", result)
self.assertIn("lxc.include: /etc/pve/proxmenux/140.sysctls\n", result)
# Proxmox renames the volumes; anything else that happens to read 119 stays.
self.assertIn("rootfs: local-lvm:vm-119-disk-0,size=8G\n", result)
self.assertIn("memory: 119\n", result)
self.assertTrue(result.endswith("\n"))
legacy = recovery.renumbered_lines("lxc.include: /etc/pve/lxc/119.proxmenux-sysctls\n", 119, 140)
self.assertEqual(legacy, "lxc.include: /etc/pve/proxmenux/140.sysctls\n")
def test_a_container_without_its_copy_stops_the_recovery(self):
broken = dict(entry(119, APP, record(119, APP)), copy=None, problem="no copy")
plans = self.plans([broken], {119: saved(APP, "app")})
self.assertEqual(plans[0]["blockers"], ["CT 119: no copy"])
class NetworkTests(unittest.TestCase):
def check(self, text, links=("lo", "vmbr0"), addresses=None, routes=(), defined=None, guests=None):
plan = {"restored": [entry(119, APP, record(119, APP), text)], "members": [{"vmid": 119}],
"blockers": [], "notes": [], "bridges": {}}
with patch.object(recovery, "links", return_value=set(links)), \
patch.object(recovery, "host_addresses", return_value=addresses or {}), \
patch.object(recovery, "routed_networks", return_value=list(routes)), \
patch.object(recovery, "defined_network", return_value=defined), \
patch.object(recovery, "local_guests", return_value=guests or {}):
recovery.check_networks(plan)
return plan
PRIVATE = config(APP, "net0: name=eth0,bridge=vmbr0,ip=dhcp\nnet1: name=eth1,bridge=vmbr11,ip=10.77.1.40/24\n")
def test_a_missing_private_network_is_created_with_the_same_subnet(self):
plan = self.check(self.PRIVATE)
self.assertEqual(plan["blockers"], [])
self.assertEqual(plan["bridges"], {"vmbr11": ipaddress.ip_network("10.77.1.0/24")})
def test_a_private_network_already_here_is_left_as_it_is(self):
plan = self.check(self.PRIVATE, links=("vmbr0", "vmbr11"),
addresses={"vmbr11": [ipaddress.ip_interface("10.77.1.1/24")]})
self.assertEqual((plan["blockers"], plan["bridges"]), ([], {}))
def test_the_subnet_in_use_on_this_host_cancels_the_recovery(self):
plan = self.check(self.PRIVATE, links=("vmbr0", "vmbr20"),
addresses={"vmbr20": [ipaddress.ip_interface("10.77.1.1/24")]})
self.assertEqual(len(plan["blockers"]), 1)
self.assertIn("10.77.1.0/24", plan["blockers"][0])
routed = self.check(self.PRIVATE, routes=[("vmbr0", ipaddress.ip_network("10.0.0.0/8"))])
self.assertEqual(len(routed["blockers"]), 1)
def test_the_bridge_name_used_for_another_network_cancels_the_recovery(self):
plan = self.check(self.PRIVATE, links=("vmbr0", "vmbr11"),
addresses={"vmbr11": [ipaddress.ip_interface("192.168.5.1/24")]})
self.assertEqual(len(plan["blockers"]), 1)
defined = self.check(self.PRIVATE, defined=ipaddress.ip_network("192.168.5.0/24"))
self.assertEqual(len(defined["blockers"]), 1)
def test_an_address_taken_by_another_container_cancels_the_recovery(self):
plan = self.check(self.PRIVATE, guests={200: "net0: name=eth0,bridge=vmbr30,ip=10.77.1.40/24\n"})
self.assertTrue(any("10.77.1.40" in line for line in plan["blockers"]))
def test_a_missing_bridge_of_the_local_network_is_not_invented(self):
plan = self.check(config(APP, "net0: name=eth0,bridge=vmbr5,ip=dhcp\n"))
self.assertEqual(len(plan["blockers"]), 1)
self.assertEqual(plan["bridges"], {})
class RecordTests(unittest.TestCase):
def test_what_a_restore_changes(self):
related = recovery.restore_related
self.assertTrue(related("rootfs", "local-lvm:vm-119-disk-0,size=8G", "tank:subvol-119-disk-0,size=8G"))
self.assertTrue(related("mp0", "local-lvm:vm-119-disk-1,mp=/data,backup=1,size=2G",
"tank:subvol-119-disk-1,mp=/data,backup=1,size=2G"))
self.assertFalse(related("mp0", "local-lvm:vm-119-disk-1,mp=/data,backup=1,size=2G",
"local-lvm:vm-119-disk-1,mp=/other,backup=1,size=2G"))
self.assertFalse(related("mp0", "/srv/a,mp=/data", "/srv/b,mp=/data"))
self.assertTrue(related("net0", "name=eth0,bridge=vmbr0,hwaddr=AA:AA,ip=dhcp", "name=eth0,bridge=vmbr0,hwaddr=BB:BB,ip=dhcp"))
self.assertFalse(related("net0", "name=eth0,bridge=vmbr0,ip=dhcp", "name=eth0,bridge=vmbr1,ip=dhcp"))
self.assertTrue(related("hookscript", "local:snippets/a.sh", "nas:snippets/a.sh"))
self.assertTrue(related("dev0", "path=/dev/dri/renderD128", None))
self.assertTrue(related("mp1", "/srv/media,mp=/media", None))
self.assertFalse(related("mp1", "local-lvm:vm-119-disk-1,mp=/data,backup=1,size=2G", None))
self.assertFalse(related("memory", "2048", "4096"))
self.assertFalse(related("mp2", None, "/etc,mp=/host"))
def standalone(self, text, value=None):
value = value or record(119, APP)
observed = {"config": text, "config_sha256": "new", "api_config_sha256": "new"}
with patch.object(instances, "observe", return_value=observed) as observe, \
patch.object(recovery, "capture_sources", return_value={}), \
patch.object(recovery, "capture_gpu_devices", return_value={}):
return recovery.standalone_record(entry(119, APP, value, text)), observe
def test_volumes_on_another_storage_become_the_reference(self):
text = config(APP).replace("local-lvm:vm-119-disk-0", "tank:subvol-119-disk-0")
result, observe = self.standalone(text)
observe.assert_called_once()
self.assertEqual(result["observed"]["config"], text)
self.assertEqual(result["deployment"]["rootfs"]["storage"], "tank")
def test_a_change_made_outside_stays_a_difference(self):
text = config(APP).replace("local-lvm:vm-119-disk-0", "tank:subvol-119-disk-0").replace("memory: 2048", "memory: 4096")
result, observe = self.standalone(text)
observe.assert_not_called()
self.assertIn("memory: 2048", result["observed"]["config"])
self.assertIn("rootfs: tank:subvol-119-disk-0,size=8G", result["observed"]["config"])
self.assertNotIn("api_config_sha256", result["observed"])
self.assertEqual(result["deployment"]["rootfs"]["storage"], "tank")
def test_a_single_container_restored_with_another_id_is_registered_under_it(self):
value = record(119, APP)
value["observed"]["config"] = config(APP, CONSOLE.format(119))
found = entry(141, APP, value, config(APP, CONSOLE.format(141)).replace("vm-119-disk-0", "vm-141-disk-0"))
with patch.object(recovery, "local_guests", return_value={141: saved(APP, "app")}), \
patch.object(recovery, "registered_identities", return_value={}):
plan = recovery.applications(Path("/nonexistent"), [found])[0]
self.assertEqual((plan["blockers"], plan["renumbered"]), ([], {119: 141}))
observed = {"config": found["config"], "config_sha256": "new"}
with patch.object(instances, "observe", return_value=observed) as observe:
result = recovery.standalone_record(plan["restored"][0])
observe.assert_called_once()
self.assertEqual((result["vmid"], result["deployment"]["vmid"]), (141, 141))
def test_the_copy_cannot_add_a_host_directory_or_a_device(self):
value = record(119, APP)
value["deployment"]["mounts"] = [
{"type": "host-bind", "source": "/etc", "container_path": "/host"},
{"type": "host-bind", "source": "/srv/media", "container_path": "/media"},
{"type": "managed-volume", "source": "local-lvm", "container_path": "/data"}]
value["deployment"]["devices"] = [{"kind": "character-device", "host_path": "/dev/kvm"},
{"kind": "character-device", "host_path": "/dev/dri/renderD128"}]
extra = "mp0: /srv/media,mp=/media,backup=0\nmp1: tank:subvol-119-disk-1,mp=/data,backup=1,size=2G\ndev0: path=/dev/dri/renderD128,gid=104\n"
value["observed"]["config"] = config(APP, extra.replace("tank:subvol", "local-lvm:vm"))
result, _ = self.standalone(config(APP, extra), value)
self.assertEqual([m["source"] for m in result["deployment"]["mounts"]], ["/srv/media", "tank"])
self.assertEqual([d["host_path"] for d in result["deployment"]["devices"]], ["/dev/dri/renderD128"])
def test_the_start_order_must_name_containers_of_the_application(self):
contract = {"schema": 1, "stack": "tandoor", "dependencies": [
{"vmid": 120, "label": "PostgreSQL",
"healthcheck": {"type": "exec", "timeout_seconds": 120, "argv": ["pg_isready"]}}]}
self.assertTrue(recovery.valid_contract(contract, {120}))
self.assertFalse(recovery.valid_contract(contract, {121}))
self.assertFalse(recovery.valid_contract(None, {120}))
http = copy.deepcopy(contract)
http["dependencies"][0]["healthcheck"] = {"type": "http", "timeout_seconds": 60, "url": "http://10.77.1.41:8080/health"}
self.assertTrue(recovery.valid_contract(http, {120}))
http["dependencies"][0]["healthcheck"]["url"] = "http://192.168.0.1/admin"
self.assertFalse(recovery.valid_contract(http, {120}))
class RcloneMountTests(unittest.TestCase):
MOUNT = {"mount_name": "drive", "shared_mount_root": "/mnt/oci-shared/remotes",
"shared_mount_read_only_root": "/mnt/oci-shared/remotes-ro", "shared_mount_root_parent": "/mnt/oci-shared"}
def test_a_mount_inside_its_common_root_is_published_again(self):
self.assertEqual(recovery.rclone_mount(dict(self.MOUNT, extra="ignored")), self.MOUNT)
def test_a_mount_the_host_cannot_publish_safely_is_left_to_the_user(self):
for change in ({"mount_name": "../etc"}, {"mount_name": "a b"}, {"shared_mount_root": "/srv/other"},
{"shared_mount_root_parent": "/etc", "shared_mount_root": "/etc/remotes",
"shared_mount_read_only_root": "/etc/remotes-ro"},
{"shared_mount_root": "/mnt/oci-shared/../../etc"}, {"shared_mount_root_parent": "/"},
{"shared_mount_read_only_root": 5}):
self.assertIsNone(recovery.rclone_mount({**self.MOUNT, **change}), change)
self.assertIsNone(recovery.rclone_mount(None))
def test_the_hookscript_of_the_mount_follows_the_new_id(self):
text = "hookscript: local:snippets/proxmenux-rclone-119-fuse-hook.sh\nmemory: 119\n"
self.assertEqual(recovery.renumbered_lines(text, 119, 143),
"hookscript: local:snippets/proxmenux-rclone-143-fuse-hook.sh\nmemory: 119\n")
other = "hookscript: local:snippets/proxmenux-stack-dependencies.sh\n"
self.assertEqual(recovery.renumbered_lines(other, 119, 143), other)
def test_the_parameters_are_read_from_the_hookscript_of_the_host(self):
hook = ("inside=/data/mounts/drive\npublished=/mnt/oci-shared/remotes/drive\n"
"published_ro=/mnt/oci-shared/remotes-ro/drive\n if ! mountpoint -q /mnt/oci-shared; then\n")
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
(Path(tmp.name) / "proxmenux-rclone-119-fuse-hook.sh").write_text(hook)
config = "arch: amd64\nhookscript: local:snippets/proxmenux-rclone-119-fuse-hook.sh\n"
with patch.object(carried, "SNIPPETS", Path(tmp.name)):
self.assertEqual(carried.rclone_mount(config), self.MOUNT)
self.assertIsNone(carried.rclone_mount("arch: amd64\n"))
self.assertIsNone(carried.rclone_mount(config.replace("119", "120")))
class NvidiaStaticTests(unittest.TestCase):
"""A privileged container names every file of the NVIDIA driver of its host."""
@staticmethod
def inventory(version, uuid):
library = f"/usr/lib/x86_64-linux-gnu/libnvidia-ml.so.{version}"
return {"gpus": [f"Quadro P1000, {uuid}, {version}"], "toolkit_version": ["cli-version: 1", "lib-version: 1"],
"devices": {"/dev/nvidia0": {"source": "/dev/nvidia0", "uid": 0, "gid": 0, "mode": 0o666,
"major": 195, "minor": 0}},
"files": {library: {"source": library, "uid": 0, "gid": 0, "mode": 0o644, "size": 1, "sha256": "x"}},
"links": {}}
def test_the_driver_files_are_named_again_for_the_host_the_container_is_on(self):
import oci_nvidia_runtime as runtime
old, new = self.inventory("550.1", "GPU-aaa"), self.inventory("560.2", "GPU-aaa")
text = ("arch: amd64\ndev0: path=/dev/nvidia0,mode=0666,gid=0,deny-write=0\n"
"lxc.mount.entry: /usr/lib/x86_64-linux-gnu/libnvidia-ml.so.550.1 "
"usr/lib/x86_64-linux-gnu/libnvidia-ml.so.550.1 none ro,bind,create=file 0 0\n").encode()
plan = runtime.refresh_plan(text, old, new)
self.assertTrue(plan["changed"])
self.assertIn(b"libnvidia-ml.so.560.2 usr/lib/x86_64-linux-gnu/libnvidia-ml.so.560.2", plan["config"])
self.assertNotIn(b"550.1", plan["config"])
other = self.inventory("560.2", "GPU-bbb")
with self.assertRaises(ValueError):
runtime.refresh_plan(text, old, other)
self.assertTrue(runtime.refresh_plan(text, old, other, same_gpu=False)["changed"])
self.assertFalse(runtime.refresh_plan(text, old, old)["changed"])
class ReturnedContainerTests(unittest.TestCase):
"""A container that came back from another node, an older backup or a
snapshot carries the record that describes it."""
def copy(self, value, generation):
return {"schema_version": 1, "kind": carried.KIND, "node": "other", "vmid": 119,
"generation": generation, "record": value}
def test_a_copy_this_host_did_not_write_with_another_record_wins(self):
host, older = record(119, APP), record(119, APP)
older["deployment"]["hostname"] = "before-the-update"
stamp = {"digest": "x", "generation": "ours"}
self.assertTrue(carried.superseded(host, self.copy(older, "theirs"), stamp))
# Written by this host: the record changed here afterwards.
self.assertFalse(carried.superseded(host, self.copy(older, "ours"), stamp))
# Came back without changes: the same record.
self.assertFalse(carried.superseded(host, self.copy(record(119, APP), "theirs"), stamp))
# Nothing was written by this host yet, or the copy is of another installation.
self.assertFalse(carried.superseded(host, self.copy(older, "theirs"), {}))
self.assertFalse(carried.superseded(host, self.copy(record(119, DB), "theirs"), stamp))
def test_the_mark_of_the_last_copy_is_kept_on_the_host(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
path = Path(tmp.name) / carried.STAMP
self.assertEqual(carried.read_stamp(path), {})
path.write_text("abc\n")
self.assertEqual(carried.read_stamp(path), {"digest": "abc"})
path.write_text('{"digest": "abc", "generation": "g1"}\n')
self.assertEqual(carried.read_stamp(path), {"digest": "abc", "generation": "g1"})
class ClusterCopyTests(unittest.TestCase):
"""Every node of a cluster reads the copy kept in /etc/pve."""
def setUp(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
self.root = Path(tmp.name)
patcher = patch.object(carried, "CLUSTER", self.root / "cluster")
patcher.start()
self.addCleanup(patcher.stop)
def copy(self, value, generation="g1", node="pve1", saved_at="2026-10-03T10:00:00+00:00"):
return {"schema_version": 1, "kind": carried.KIND, "node": node, "vmid": value["vmid"],
"generation": generation, "saved_at": saved_at, "record": value}
def test_the_copy_is_written_and_read_back(self):
self.assertIsNone(carried.read_cluster(119))
self.assertTrue(carried.write_cluster(119, self.copy(record(119, APP))))
self.assertEqual(carried.read_cluster(119)["record"]["installation_id"], APP)
(carried.CLUSTER / "120.json").write_text("not json")
self.assertIsNone(carried.read_cluster(120))
carried.remove_cluster(119)
self.assertIsNone(carried.read_cluster(119))
def test_it_follows_the_copy_the_container_carries(self):
value = self.copy(record(119, APP))
carried.keep_cluster(119, value, "g1")
written = (carried.CLUSTER / "119.json").stat().st_mtime_ns
carried.keep_cluster(119, value, "g1")
self.assertEqual((carried.CLUSTER / "119.json").stat().st_mtime_ns, written)
carried.keep_cluster(119, value, "g2")
self.assertEqual(carried.read_cluster(119)["generation"], "g2")
def test_a_record_changed_on_another_node_replaces_the_one_of_this_host(self):
path = self.root / "oci-compose.json"
path.write_text("{}")
os.utime(path, (1_000_000_000, 1_000_000_000))
host, changed = record(119, APP), record(119, APP)
changed["deployment"]["hostname"] = "updated-on-the-other-node"
self.assertTrue(carried.replaced_elsewhere(host, path, self.copy(changed, node="pve2"), "pve1"))
# Written by this node, the same record, another installation or an older copy.
self.assertFalse(carried.replaced_elsewhere(host, path, self.copy(changed, node="pve1"), "pve1"))
self.assertFalse(carried.replaced_elsewhere(host, path, self.copy(record(119, APP), node="pve2"), "pve1"))
self.assertFalse(carried.replaced_elsewhere(host, path, self.copy(record(119, DB), node="pve2"), "pve1"))
old = self.copy(changed, node="pve2", saved_at="1999-01-01T00:00:00+00:00")
self.assertFalse(carried.replaced_elsewhere(host, path, old, "pve1"))
self.assertFalse(carried.replaced_elsewhere(host, path, None, "pve1"))
def examine(self, inside, shared):
if shared is not None:
carried.write_cluster(119, shared)
done = type("Done", (), {"returncode": 0, "stdout": config(APP)})()
class Root:
def __enter__(self): return Path("/rootfs")
def __exit__(self, *_): return False
with patch.object(recovery, "run", return_value=done), \
patch.object(carried, "container_root", return_value=Root()), \
patch.object(carried, "read_copy", return_value=inside):
return recovery.examine({"vmid": 119, "installation_id": APP, "hostname": "app"})
def test_the_same_copy_in_the_cluster_is_taken_without_asking(self):
value = self.copy(record(119, APP), "g1")
result = self.examine(dict(value), value)
self.assertTrue(result["trusted"])
self.assertIsNone(result["problem"])
def test_a_container_of_another_moment_is_described_by_its_own_copy(self):
older = record(119, APP)
older["deployment"]["hostname"] = "before"
result = self.examine(self.copy(older, "g0"), self.copy(record(119, APP), "g1"))
self.assertFalse(result["trusted"])
self.assertEqual(result["copy"]["record"]["deployment"]["hostname"], "before")
def test_a_container_without_its_copy_takes_the_one_of_the_cluster_and_asks(self):
result = self.examine(None, self.copy(record(119, APP), "g1"))
self.assertFalse(result["trusted"])
self.assertIsNone(result["problem"])
self.assertIsNotNone(result["copy"])
carried.remove_cluster(119)
nothing = self.examine(None, None)
self.assertTrue(nothing["unrecoverable"])
def test_the_copy_of_another_installation_is_not_used(self):
carried.remove_cluster(119)
result = self.examine(self.copy(record(119, APP), "g1"), self.copy(record(119, DB), "g1"))
self.assertFalse(result["trusted"])
self.assertEqual(result["copy"]["record"]["installation_id"], APP)
def test_only_what_needs_no_question_is_registered_on_its_own(self):
def plan(**extra):
item = dict(entry(119, APP, record(119, APP)), trusted=True)
item.update(extra.pop("entry", {}))
return {"blockers": [], "bridges": {}, "renumbered": {}, "hold": False, "restored": [item], **extra}
self.assertTrue(recovery.automatic(plan()))
self.assertFalse(recovery.automatic(plan(entry={"trusted": False})))
self.assertFalse(recovery.automatic(plan(bridges={"vmbr11": "10.77.1.0/24"})))
self.assertFalse(recovery.automatic(plan(blockers=["missing"])))
self.assertFalse(recovery.automatic(plan(renumbered={119: 140})))
self.assertFalse(recovery.automatic(plan(entry={"firewall": {"port": 19999}})))
self.assertFalse(recovery.automatic(plan(entry={"contract": {"schema": 1}})))
class RegistrationTests(unittest.TestCase):
def test_no_record_is_left_when_the_application_cannot_be_registered_whole(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
root = Path(tmp.name)
database = record(120, DB, stack_member={"stack_id": APP, "primary_vmid": 119, "name": "database"})
application = record(119, APP, stack_member={"stack_id": APP, "primary_vmid": 119, "name": "application"})
plan = {"restored": [entry(119, APP, application), entry(120, DB, database)]}
def refuse(_root, vmid):
if vmid == 120:
raise ValueError("cannot be updated")
with patch.object(instances, "private_directory", lambda path: path.mkdir(parents=True, exist_ok=True)), \
patch.object(recovery.oci_stack_modify, "register", side_effect=refuse):
with self.assertRaises(ValueError):
recovery.register(root, plan)
self.assertFalse(instances.location(root, 119).exists())
self.assertFalse(instances.location(root, 120).exists())
def test_the_plan_each_container_was_installed_with_is_left_as_it_is_now(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
root = Path(tmp.name)
database = record(150, DB, stack_member={"stack_id": APP, "primary_vmid": 151, "name": "database"})
database["deployment"]["mounts"] = [{"source": "local-lvm:vm-150-disk-1"}]
application = record(151, APP, stack_member={"stack_id": APP, "primary_vmid": 151, "name": "application"})
application["stack"] = {"id": APP, "members": [], "deployment": {"services": [
{"name": "database", "vmid": 150, "deployment": {"mounts": [{"source": "local-lvm:vm-120-disk-1"}]}},
{"name": "application", "vmid": 151, "deployment": {"create_arguments": ["119"]}},
{"name": "check", "vmid": 150, "healthcheck": {"type": "running"}}]}}
with patch.object(instances, "private_directory", lambda path: path.mkdir(parents=True, exist_ok=True)):
instances.write(instances.location(root, 150), database)
instances.write(instances.location(root, 151), application)
recovery.refresh_service_plans(root, [151, 150])
services = instances.read(root, 151)["stack"]["deployment"]["services"]
self.assertEqual(services[0]["deployment"]["mounts"], [{"source": "local-lvm:vm-150-disk-1"}])
self.assertEqual(services[1]["deployment"], application["deployment"])
self.assertNotIn("deployment", services[2])
if __name__ == "__main__":
unittest.main()
+47
View File
@@ -0,0 +1,47 @@
"""An image is the same in the registry and in the archive made from it,
whichever format the registry publishes its manifest in."""
from pathlib import Path
import sys
import unittest
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
from oci_installation_state import same_image
REGISTRY = "sha256:" + "a" * 64
ARCHIVE = "sha256:" + "b" * 64
LAYERS = ["sha256:" + "c" * 64, "sha256:" + "d" * 64]
BUILT = "2026-09-15T01:13:55.019799592Z"
class SameImageTests(unittest.TestCase):
def test_an_oci_manifest_keeps_its_digest(self):
self.assertTrue(same_image({"manifest_digest": REGISTRY}, {"manifest_digest": REGISTRY}))
def test_a_docker_manifest_is_recognised_by_its_layers_and_when_it_was_built(self):
self.assertTrue(same_image({"manifest_digest": REGISTRY, "layers": LAYERS, "created": BUILT},
{"manifest_digest": ARCHIVE, "layers": list(LAYERS), "created": BUILT}))
def test_another_image_is_not_the_same(self):
candidate = {"manifest_digest": REGISTRY, "layers": LAYERS, "created": BUILT}
self.assertFalse(same_image(candidate, {"manifest_digest": ARCHIVE, "layers": LAYERS[:1], "created": BUILT}))
self.assertFalse(same_image(candidate, {"manifest_digest": ARCHIVE, "layers": LAYERS[::-1], "created": BUILT}))
# The same layers rebuilt with another configuration are another image.
self.assertFalse(same_image(candidate, {"manifest_digest": ARCHIVE, "layers": LAYERS,
"created": "2026-10-01T00:00:00Z"}))
def test_what_is_missing_never_matches(self):
self.assertFalse(same_image({"manifest_digest": REGISTRY}, {"manifest_digest": ARCHIVE}))
self.assertFalse(same_image({"manifest_digest": REGISTRY, "layers": [], "created": BUILT},
{"manifest_digest": ARCHIVE, "layers": [], "created": BUILT}))
self.assertFalse(same_image({"manifest_digest": REGISTRY, "layers": LAYERS, "created": None},
{"manifest_digest": ARCHIVE, "layers": LAYERS, "created": None}))
# A record saved before the layers were kept is compared by its digest alone.
self.assertFalse(same_image({"manifest_digest": REGISTRY, "layers": LAYERS, "created": BUILT},
{"manifest_digest": ARCHIVE}))
if __name__ == "__main__":
unittest.main()
+47
View File
@@ -0,0 +1,47 @@
"""The notes Proxmox shows for a multi-container application link to the
address it answers on from the local network, not to its private leg."""
from pathlib import Path
import sys
from types import SimpleNamespace
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
import oci_native_stack as stack
TWO_LEGS = "net0: name=eth0,bridge=vmbr0,ip=dhcp,type=veth\nnet1: name=eth1,bridge=vmbr10,ip=10.77.0.10/24,type=veth\n"
PRIVATE_ONLY = "net0: name=eth0,bridge=vmbr10,ip=10.77.0.12/24,type=veth\n"
class AccessAddressTests(unittest.TestCase):
def address(self, config, *answers, wait=20):
listed = iter(answers)
def run(command, **_):
if command[0] == "pct":
return SimpleNamespace(stdout=config)
return SimpleNamespace(stdout=next(listed))
with patch.object(stack.subprocess, "run", side_effect=run), patch.object(stack.time, "sleep"):
return stack.access_address(101, wait)
def test_the_address_of_the_local_network_is_used(self):
self.assertEqual(self.address(TWO_LEGS, "10.77.0.10\n192.168.0.44\n"), "192.168.0.44")
self.assertEqual(self.address(TWO_LEGS, "192.168.0.44\n10.77.0.10\n"), "192.168.0.44")
def test_a_lease_that_has_not_arrived_is_waited_for(self):
self.assertEqual(self.address(TWO_LEGS, "10.77.0.10\n", "10.77.0.10\n192.168.0.44\n"), "192.168.0.44")
def test_a_container_with_only_its_private_leg_keeps_that_address(self):
self.assertEqual(self.address(PRIVATE_ONLY, "10.77.0.12\n"), "10.77.0.12")
def test_a_container_without_address_has_none(self):
self.assertEqual(self.address(PRIVATE_ONLY, "\n"), "")
self.assertEqual(self.address(TWO_LEGS, "10.77.0.10\n", wait=0), "10.77.0.10")
if __name__ == "__main__":
unittest.main()
+62
View File
@@ -0,0 +1,62 @@
"""The temporary container that holds the data of an application during an
update is removed when the operation ends: empty after a commit, and with the
disks of the failed attempt after a verified recovery."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
import oci_instance_transaction as transaction
EMPTY = b"arch: amd64\nhostname: oci-data-holder\nrootfs: local-lvm:vm-104-disk-0,size=8G\n"
HOLDING = EMPTY + b"mp0: local-lvm:vm-104-disk-1,mp=/transaction-retained/bd021f9e,backup=1,size=32G\n"
STATE = {"id": "4dfbfd9223a440c1b07a0334b08fc4fd", "stage": 104}
class StageReleaseTests(unittest.TestCase):
def release(self, config, **arguments):
calls = []
with patch.object(transaction, "owned", return_value=config) as owned, \
patch.object(transaction, "stop", side_effect=lambda vmid: calls.append(("stop", vmid))), \
patch.object(transaction, "run", side_effect=lambda *command: calls.append(command)):
transaction.release_stage(STATE, **arguments)
owned.assert_called_once_with(104, "proxmenux-transaction=" + STATE["id"])
return calls
def test_after_a_commit_the_empty_container_is_removed(self):
self.assertEqual(self.release(EMPTY), [("pct", "destroy", "104")])
def test_a_container_that_still_holds_a_disk_is_kept(self):
self.assertEqual(self.release(HOLDING), [])
self.assertEqual(self.release(EMPTY + b"unused0: local-lvm:vm-104-disk-2\n"), [])
def test_after_a_verified_recovery_it_is_removed_with_what_it_holds(self):
self.assertEqual(self.release(HOLDING, discard=True), [("stop", 104), ("pct", "destroy", "104")])
def test_a_container_of_another_operation_is_never_removed(self):
calls = []
with patch.object(transaction, "owned", side_effect=ValueError("not ours")), \
patch.object(transaction, "run", side_effect=lambda *command: calls.append(command)):
transaction.release_stage(STATE, discard=True)
self.assertEqual(calls, [])
def test_an_operation_without_a_temporary_container_does_nothing(self):
with patch.object(transaction, "owned") as owned:
transaction.release_stage({"id": "x"}, discard=True)
owned.assert_not_called()
def test_a_removal_that_fails_does_not_undo_the_recovery(self):
with patch.object(transaction, "owned", return_value=HOLDING), \
patch.object(transaction, "stop"), \
patch.object(transaction, "run", side_effect=RuntimeError("pct failed with exit code 255")), \
patch.object(transaction, "log") as logged:
transaction.discard_stage(STATE)
self.assertIn("pct failed", logged.call_args[0][0])
if __name__ == "__main__":
unittest.main()
+8 -3
View File
@@ -1,6 +1,6 @@
#!/bin/bash
# ==========================================================
# ProxMenux - Update or recreate one OCI instance
# ProxMenux - Update, recreate or recover OCI instances
# ==========================================================
# Author : MacRimi
# Copyright : (c) 2024 MacRimi
@@ -13,8 +13,10 @@
# OCI manager Apps -> Manage installed OCI applications for one
# container, without the list:
#
# VMID - the container (required)
# ACTION - "update" or "recreate" (required)
# VMID - the container (required for update and recreate)
# ACTION - "update", "recreate" or "recover" (required);
# "recover" registers again the applications
# restored from a backup
# KEEP_BACKUP - storage where the backup taken before the
# update is kept (optional)
# ==========================================================
@@ -29,6 +31,9 @@ fi
load_language
initialize_cache
if [[ ${ACTION:-} == "recover" ]]; then
exec bash "$LOCAL_SCRIPTS/oci/oci_manager_apps.sh" recover
fi
if [[ ! ${VMID:-} =~ ^[0-9]{1,9}$ ]]; then
msg_error "$(translate "Invalid VMID")"
exit 1
@@ -23,5 +23,5 @@ export async function generateMetadata({ params }: { params: Promise<{ locale: s
export default async function Page({ params }: { params: Promise<{ locale: string }> }) {
const { locale } = await params
setRequestLocale(locale)
return <DocPage locale={locale} namespace={NAMESPACE} minutes={7} links={LINKS} />
return <DocPage locale={locale} namespace={NAMESPACE} minutes={9} links={LINKS} />
}
@@ -61,6 +61,7 @@
"rows": [
["Intel/AMD DRM", "<code>/dev/dri/renderD*</code> and <code>/sys/class/drm/NODE/device/vendor</code>", "A character device with vendor <code>0x8086</code> (Intel) or <code>0x1002</code> (AMD)"],
["AMD OpenCL", "The render node, plus <code>/dev/kfd</code> when the profile needs it", "Existence, type, vendor, permissions and declared compatibility"],
["AMD ROCm", "The render node, <code>/dev/kfd</code> and the generation the compute driver reports in <code>/sys/class/kfd/kfd/topology/nodes/*/properties</code>", "A generation the ROCm image of the application carries kernels for, the RDNA2 GPUs it names and newer ones, is offered as it is. The other GPUs of those families, such as the Radeon 680M and 780M, are offered as experimental and are never the proposed option: ROCm does not support them officially, the application presents them as the generation of their family (<code>HSA_OVERRIDE_GFX_VERSION</code>) and the installer asks for confirmation before the larger image is downloaded. An older GPU, such as the integrated graphics of a Ryzen 5000U, is not offered and Immich keeps recognition on the CPU"],
["NVIDIA", "<code>nvidia-smi</code> and <code>nvidia-container-cli</code>", "GPU, UUID, PCI bus, driver version, Toolkit, <code>/dev/nvidia*</code> nodes, binaries and libraries"],
["Coral PCIe/M.2", "<code>/dev/apex_N</code> and its link in <code>/sys/dev/char/MAJOR:MINOR</code>", "Character node, major/minor, owner, GID and permissions"],
["USB and serial", "<code>/dev/ttyUSB*</code>, <code>/dev/ttyACM*</code> or <code>/dev/bus/usb/BBB/DDD</code>", "Character node; for USB also vendor, product and serial when sysfs publishes them"],
+205 -4
View File
@@ -1,7 +1,7 @@
{
"meta": {
"title": "Install, update and recreate | ProxMenux",
"description": "The instance contract of an OCI container and the operations that use it: update, recreate, remove and recovery of an interrupted operation."
"description": "The instance contract of an OCI container and the operations that use it: update, recreate, remove, recovery of an interrupted operation and recovery after a restore on this or another host."
},
"header": {
"title": "Install, update and recreate",
@@ -20,6 +20,9 @@
"code": {
"code": "instances/\n└── 105/\n └── oci-compose.json\n ├── image and resolved digest\n ├── resources and network\n ├── environment (secrets protected)\n ├── container disks and host directories\n ├── hardware profile and devices\n ├── console log and terminal mode\n └── stack membership and lifecycle"
}
},
{
"p": "A copy of the contract travels inside the container, in <code>/.proxmenux/oci-record.json</code>, readable only by root of the container, so a backup of the container always carries the contract it had at that moment. A second copy is kept in <code>/etc/pve/priv/proxmenux/oci</code>, which every node of a cluster shares and only root of the host reads. Both are written after every installation, update and recreation."
}
]
},
@@ -55,7 +58,7 @@
}
},
{
"p": "For a multi-container application the menu offers <strong>Update every container of the application</strong> and the removal. A stack is not recreated."
"p": "For a multi-container application the menu offers <strong>Update every container of the application</strong>, <strong>Recreate</strong> and the removal. Recreate adds or removes the extra paths and devices of the application container without rebuilding anything and, in Immich, changes what runs recognition."
},
{
"figure": {
@@ -155,6 +158,138 @@
}
]
},
{
"id": "restore",
"title": "Backup, restore and another host",
"intro": "A backup made with vzdump or Proxmox Backup Server includes the container, its disks and the copy of its contract. What the installation keeps on the host is not part of it. <strong>Manage installed OCI applications</strong> detects the containers restored on a host that has no contract for them, a new Proxmox installation or another host, and offers to register them again.",
"blocks": [
{
"steps": {
"items": [
{
"title": "Restore the containers in Proxmox",
"body": "From the backup storage, with the original ID or with any free one. A multi-container application needs every one of its containers."
},
{
"title": "Open Manage installed OCI applications",
"body": "The restored containers are listed and the recovery is offered. The <strong>Recover</strong> button of the Updates tab of <monitorLink>ProxMenux Monitor</monitorLink> opens the same recovery."
},
{
"title": "Check before changing",
"body": "The recovery checks that the host has everything each application needs. An application that cannot be recovered whole is left as it was found, with the reason."
},
{
"title": "Register and start",
"body": "The contract is registered on this host and what the installation kept on it is written again. Starting the applications is a separate question, answered with No when the original containers are still running on another host."
}
]
}
},
{
"figure": {
"src": "/oci-manager/restore-offer.png",
"alt": "Dialog that lists the restored containers and offers to recover them",
"caption": "The recovery offered when Manage installed OCI applications opens."
}
},
{
"table": {
"headers": [
"What the host kept",
"After the recovery"
],
"rows": [
[
"Instance contract",
"Registered from the copy the container carries, checked against the configuration Proxmox restored"
],
[
"Private network of a multi-container application",
"Created again with the same bridge and subnet; the fixed addresses of the containers do not change"
],
[
"Start order of a multi-container application",
"The dependency hookscript and its contract are installed again; snippets are enabled on the <code>local</code> storage when no storage accepts them"
],
[
"Network sysctls and host-monitor file",
"Written again in <code>/etc/pve/proxmenux</code>"
],
[
"NVIDIA runtime",
"The hook of an unprivileged container is installed again. A privileged container gets the driver files of this host instead of those of the host it comes from"
],
[
"Rclone mount",
"The hookscript and the programs that publish the mount are written again, with the same views on the host"
],
[
"Host firewall rule of a host monitor",
"Asked again, for its web port and the subnet of the bridge on this host"
],
[
"<code>lost+found</code> of each restored disk",
"Removed when empty; a restore creates it and some applications cannot start with it in their data"
],
[
"Disks restored on another storage",
"The contract is updated to the storage they are on now"
]
]
}
},
{
"p": "A container restored with another ID keeps its application. The contract, its console log, its network sysctls, the hookscript of an Rclone mount and the start order of a multi-container application are registered with the IDs the containers have on this host."
},
{
"table": {
"headers": [
"What stops a recovery",
"What to do"
],
"rows": [
[
"A container of a multi-container application is missing",
"Restore it too; the application is recovered whole"
],
[
"The private subnet is already used on this host",
"The recovery is cancelled and nothing is changed: the addresses of the containers are fixed and are not moved to another subnet"
],
[
"A host directory does not exist",
"Mount or create it with its data; a backup of the container does not include host directories"
],
[
"A device does not exist on this host",
"Connect it, or remove it from the container in Proxmox"
],
[
"An application with NVIDIA on a host without the driver",
"Install the NVIDIA driver and the Container Toolkit"
],
[
"The backup was made before the copy of the contract existed",
"The container stays as an ordinary LXC and is not offered again"
]
]
}
},
{
"calloutInfo": {
"title": "A container that comes back",
"body": "A container restored over itself from an older backup, rolled back to a snapshot or returned from another node carries the contract of the state it is in. When it is selected for an operation and its contract differs from the one of this host, the operation does not start and the container is offered for recovery."
}
},
{
"figure": {
"src": "/oci-manager/restore-result.png",
"alt": "Result of the recovery with the private network, the start order and the registered containers",
"caption": "The result of a recovery, step by step."
}
}
]
},
{
"id": "remove",
"title": "Removing an OCI application",
@@ -180,9 +315,19 @@
"Kept, with its content",
"Other applications may use it"
],
[
"Host files of the container",
"Deleted",
"Its console log, network sysctls, Rclone mount hookscript and its registration in the App tab of ProxMenux Monitor serve no other container"
],
[
"Files several installations share",
"Deleted with the last installation that uses them",
"The host-monitor file and the dependency hookscript of multi-container applications"
],
[
"Instance contract",
"Retired after a successful removal",
"Deleted after a successful removal",
"No CT is associated with it any more"
],
[
@@ -190,10 +335,20 @@
"Released with the stack",
"It has no members left to connect"
],
[
"Network shared by the Arr suite",
"Released with the last application of the suite",
"Each application of the suite is independent and is removed on its own"
],
[
"A single member of a stack",
"Not removed on its own",
"The whole application is removed, so no stack is left incomplete"
],
[
"A container on another node of the cluster",
"Not removed",
"Its record and host files are on the node where it was installed: it is removed there, after migrating it back"
]
]
}
@@ -227,7 +382,53 @@
"title": "Registry and cleanup",
"blocks": [
{
"p": "The registered contracts are compared with the real CTs. A contract is orphaned only when its VMID no longer exists or no longer carries the expected instance identity. The cleanup does not delete volumes or external data by inference."
"p": "When <strong>Manage installed OCI applications</strong> opens, what is left of containers that exist on no node of the cluster is removed first: the saved record of a container deleted from the Proxmox interface, which is kept as history, and the host files a removal left behind. A record with an operation left halfway is kept, because its backup may still be needed. Volumes and host directories are never deleted by inference."
}
]
},
{
"id": "cluster",
"title": "In a cluster: migration and high availability",
"intro": "An OCI container is an ordinary Proxmox LXC: it can be migrated or managed by HA like any other, within the same limits. What it needs to start is kept where every node of the cluster finds it.",
"blocks": [
{
"table": {
"headers": [
"Part",
"On another node of the cluster"
],
"rows": [
[
"Console log",
"The container creates <code>/var/log/proxmenux/oci</code> before it starts, on whichever node runs it. Each node keeps the log of the starts it ran."
],
[
"Network sysctls and host monitor",
"Kept in <code>/etc/pve/proxmenux</code>, which every node of the cluster shares, so a migrated container finds them."
],
[
"Container disks",
"Proxmox moves them with the container. HA needs them on shared storage."
],
[
"Host directories and devices",
"The same rules as any LXC: a host directory must exist on the target node and be marked as shared, and a GPU, Coral or NPU must be present there."
],
[
"ProxMenux record",
"The contract stays on the node where the application was installed, and every node reads the copy kept in <code>/etc/pve/priv/proxmenux/oci</code>. A single-container application that migrates is registered on the node it arrives at when <strong>Manage installed OCI applications</strong> opens or an operation is launched for it. A multi-container application is offered for recovery there, since its private network has to be created on that node."
]
]
}
},
{
"calloutWarning": {
"title": "Not for high availability",
"body": "A multi-container application reaches its members through a private bridge of the node it was installed on, so its members stay on that node. A host monitor, such as Glances in host mode or Netdata, monitors the node it runs on; moving it would monitor another node."
}
},
{
"p": "A backup restored outside the cluster gets the files of <code>/etc/pve/proxmenux</code> the container uses when the application is recovered."
}
]
},
@@ -108,11 +108,15 @@
],
[
"Recreate",
"The recreation editor (resources, network, paths and GPU). It is not offered for a multi-container application."
"The recreation editor (resources, network, paths and GPU). In a multi-container application it adds or removes the extra paths and devices of the application container and, in Immich, changes what runs recognition."
],
[
"Recover",
"Replaces Update when an operation on the container was interrupted, and opens its recovery."
],
[
"Recover (restored container)",
"Shown for a container restored from a backup that has no contract on this host, under <strong>Restored OCI application</strong>. Opens the <lifecycleLink>recovery</lifecycleLink> in the Monitor terminal; Update and Recreate appear once it is registered."
]
]
}
+22 -1
View File
@@ -295,7 +295,7 @@
{
"id": "manage",
"title": "Managing an installed stack",
"intro": "In <strong>Manage installed OCI applications</strong>, any member leads to the whole stack. The menu of a stack offers <strong>Update every container of the application</strong> and <strong>Remove: delete the application and its containers</strong>.",
"intro": "In <strong>Manage installed OCI applications</strong>, any member leads to the whole stack. The menu of a stack offers <strong>Update every container of the application</strong>, <strong>Recreate: add or remove extra paths and devices</strong> and <strong>Remove: delete the application and its containers</strong>.",
"blocks": [
{
"steps": {
@@ -327,6 +327,27 @@
]
}
},
{
"table": {
"headers": [
"Recreate",
"What changes",
"What is kept"
],
"rows": [
[
"Extra paths and devices",
"They are added to or removed from the application container, which restarts. Nothing is rebuilt",
"The data of the application, its database and the other containers"
],
[
"What runs recognition (Immich)",
"The machine learning container takes the image and the devices of the CPU or of a GPU of the host. The whole application is stopped and updated, as in an update",
"The model cache, the library and the database. If anything fails, the previous containers and the previous choice are restored"
]
]
}
},
{
"calloutWarning": {
"title": "A stack without a coordinated replay is not updated",
@@ -61,6 +61,7 @@
"rows": [
["Intel/AMD DRM", "<code>/dev/dri/renderD*</code> y <code>/sys/class/drm/NODO/device/vendor</code>", "Un dispositivo de caracteres con fabricante <code>0x8086</code> (Intel) o <code>0x1002</code> (AMD)"],
["OpenCL AMD", "El render node, más <code>/dev/kfd</code> cuando el perfil lo necesita", "Existencia, tipo, fabricante, permisos y compatibilidad declarada"],
["AMD ROCm", "El nodo de render, <code>/dev/kfd</code> y la generación que declara el driver de cómputo en <code>/sys/class/kfd/kfd/topology/nodes/*/properties</code>", "Una generación para la que la imagen ROCm de la aplicación trae kernels, las GPU RDNA2 que incluye y las posteriores, se ofrece tal cual. Las demás GPU de esas familias, como las Radeon 680M y 780M, se ofrecen como experimentales y nunca son la opción propuesta: ROCm no las soporta oficialmente, la aplicación las presenta como la generación de su familia (<code>HSA_OVERRIDE_GFX_VERSION</code>) y el instalador pide confirmación antes de descargar la imagen, que es más grande. Una GPU anterior, como los gráficos integrados de un Ryzen 5000U, no se ofrece e Immich deja el reconocimiento en la CPU"],
["NVIDIA", "<code>nvidia-smi</code> y <code>nvidia-container-cli</code>", "GPU, UUID, bus PCI, versión del driver, Toolkit, nodos <code>/dev/nvidia*</code>, binarios y librerías"],
["Coral PCIe/M.2", "<code>/dev/apex_N</code> y su enlace en <code>/sys/dev/char/MAJOR:MINOR</code>", "Nodo de caracteres, major/minor, propietario, GID y permisos"],
["USB y serie", "<code>/dev/ttyUSB*</code>, <code>/dev/ttyACM*</code> o <code>/dev/bus/usb/BBB/DDD</code>", "Nodo de caracteres; en USB también fabricante, producto y número de serie cuando sysfs los publica"],
+205 -4
View File
@@ -1,7 +1,7 @@
{
"meta": {
"title": "Instalar, actualizar y recrear | ProxMenux",
"description": "El contrato de instancia de un contenedor OCI y las operaciones que lo usan: actualizar, recrear, eliminar y recuperar una operación interrumpida."
"description": "El contrato de instancia de un contenedor OCI y las operaciones que lo usan: actualizar, recrear, eliminar, recuperar una operación interrumpida y recuperar tras una restauración en este u otro host."
},
"header": {
"title": "Instalar, actualizar y recrear",
@@ -20,6 +20,9 @@
"code": {
"code": "instances/\n└── 105/\n └── oci-compose.json\n ├── imagen y digest resuelto\n ├── recursos y red\n ├── entorno (secretos protegidos)\n ├── discos del contenedor y directorios del host\n ├── perfil de hardware y dispositivos\n ├── log de consola y modo de terminal\n └── pertenencia a una pila y ciclo de vida"
}
},
{
"p": "Una copia del contrato viaja dentro del contenedor, en <code>/.proxmenux/oci-record.json</code>, legible solo por root del contenedor, así que un backup del contenedor lleva siempre el contrato que tenía en ese momento. Una segunda copia se guarda en <code>/etc/pve/priv/proxmenux/oci</code>, que comparten todos los nodos de un clúster y solo lee root del host. Las dos se escriben tras cada instalación, actualización y recreación."
}
]
},
@@ -55,7 +58,7 @@
}
},
{
"p": "Para una aplicación multicontenedor el menú ofrece <strong>Actualizar cada contenedor de la aplicación</strong> y la eliminación. Una pila no se recrea."
"p": "Para una aplicación multicontenedor el menú ofrece <strong>Actualizar cada contenedor de la aplicación</strong>, <strong>Recrear</strong> y la eliminación. Recrear añade o elimina las rutas y los dispositivos extra del contenedor de la aplicación sin reconstruir nada y, en Immich, cambia qué ejecuta el reconocimiento."
},
{
"figure": {
@@ -155,6 +158,138 @@
}
]
},
{
"id": "restore",
"title": "Backup, restauración y otro host",
"intro": "Un backup hecho con vzdump o Proxmox Backup Server incluye el contenedor, sus discos y la copia de su contrato. Lo que la instalación guarda en el host no forma parte de él. <strong>Gestionar aplicaciones OCI instaladas</strong> detecta los contenedores restaurados en un host que no tiene su contrato, un Proxmox recién instalado u otro host, y ofrece registrarlos de nuevo.",
"blocks": [
{
"steps": {
"items": [
{
"title": "Restaurar los contenedores en Proxmox",
"body": "Desde el almacenamiento de backups, con el ID original o con cualquiera libre. Una aplicación de varios contenedores necesita todos sus contenedores."
},
{
"title": "Abrir Gestionar aplicaciones OCI instaladas",
"body": "Se listan los contenedores restaurados y se ofrece la recuperación. El botón <strong>Recuperar</strong> de la pestaña Updates de <monitorLink>ProxMenux Monitor</monitorLink> abre la misma recuperación."
},
{
"title": "Comprobar antes de cambiar",
"body": "La recuperación comprueba que el host tiene todo lo que necesita cada aplicación. Una aplicación que no se puede recuperar entera queda como estaba, con el motivo."
},
{
"title": "Registrar y arrancar",
"body": "El contrato se registra en este host y se escribe de nuevo lo que la instalación guardaba en él. Arrancar las aplicaciones es una pregunta aparte, que se responde con No cuando los contenedores originales siguen en marcha en otro host."
}
]
}
},
{
"figure": {
"src": "/oci-manager/restore-offer.png",
"alt": "Diálogo que lista los contenedores restaurados y ofrece recuperarlos",
"caption": "La recuperación que se ofrece al abrir Gestionar aplicaciones OCI instaladas."
}
},
{
"table": {
"headers": [
"Lo que guardaba el host",
"Tras la recuperación"
],
"rows": [
[
"Contrato de la instancia",
"Se registra desde la copia que lleva el contenedor, comprobada contra la configuración que restauró Proxmox"
],
[
"Red privada de una aplicación de varios contenedores",
"Se crea de nuevo con el mismo bridge y la misma subred; las direcciones fijas de los contenedores no cambian"
],
[
"Orden de arranque de una aplicación de varios contenedores",
"El hookscript de dependencias y su contrato se instalan de nuevo; se activan los snippets en el almacenamiento <code>local</code> cuando ningún almacenamiento los admite"
],
[
"Sysctl de red y archivo del monitor del host",
"Se escriben de nuevo en <code>/etc/pve/proxmenux</code>"
],
[
"Runtime de NVIDIA",
"El hook de un contenedor sin privilegios se instala de nuevo. Un contenedor privilegiado recibe los archivos del driver de este host en lugar de los del host del que viene"
],
[
"Montaje de Rclone",
"El hookscript y los programas que publican el montaje se escriben de nuevo, con las mismas vistas en el host"
],
[
"Regla del firewall del host de un monitor del host",
"Se pregunta de nuevo, para su puerto web y la subred del bridge en este host"
],
[
"<code>lost+found</code> de cada disco restaurado",
"Se elimina cuando está vacío; una restauración lo crea y algunas aplicaciones no arrancan con él entre sus datos"
],
[
"Discos restaurados en otro almacenamiento",
"El contrato se actualiza al almacenamiento en el que están ahora"
]
]
}
},
{
"p": "Un contenedor restaurado con otro ID conserva su aplicación. El contrato, su log de consola, sus sysctl de red, el hookscript de un montaje de Rclone y el orden de arranque de una aplicación de varios contenedores se registran con los ID que tienen los contenedores en este host."
},
{
"table": {
"headers": [
"Qué detiene una recuperación",
"Qué hacer"
],
"rows": [
[
"Falta un contenedor de una aplicación de varios contenedores",
"Restaurarlo también; la aplicación se recupera entera"
],
[
"La subred privada ya se usa en este host",
"La recuperación se cancela y no se cambia nada: las direcciones de los contenedores son fijas y no se mueven a otra subred"
],
[
"Un directorio del host no existe",
"Montarlo o crearlo con sus datos; un backup del contenedor no incluye los directorios del host"
],
[
"Un dispositivo no existe en este host",
"Conectarlo, o eliminarlo del contenedor en Proxmox"
],
[
"Una aplicación con NVIDIA en un host sin el driver",
"Instalar el driver de NVIDIA y el Container Toolkit"
],
[
"El backup es anterior a que existiera la copia del contrato",
"El contenedor queda como un LXC normal y no se vuelve a ofrecer"
]
]
}
},
{
"calloutInfo": {
"title": "Un contenedor que vuelve",
"body": "Un contenedor restaurado sobre sí mismo desde un backup anterior, devuelto a un snapshot o llegado de vuelta desde otro nodo lleva el contrato del estado en el que está. Al seleccionarlo para una operación, si su contrato es distinto del de este host, la operación no empieza y el contenedor se ofrece para recuperarlo."
}
},
{
"figure": {
"src": "/oci-manager/restore-result.png",
"alt": "Resultado de la recuperación con la red privada, el orden de arranque y los contenedores registrados",
"caption": "El resultado de una recuperación, paso a paso."
}
}
]
},
{
"id": "remove",
"title": "Eliminar una aplicación OCI",
@@ -180,9 +315,19 @@
"Se conserva, con su contenido",
"Otras aplicaciones pueden usarlo"
],
[
"Archivos del contenedor en el host",
"Se eliminan",
"Su log de consola, los sysctl de red, el hookscript de montaje del Rclone y su registro en la pestaña App de ProxMenux Monitor no sirven a ningún otro contenedor"
],
[
"Archivos que comparten varias instalaciones",
"Se eliminan con la última instalación que los usa",
"El archivo del monitor del host y el hookscript de dependencias de las aplicaciones de varios contenedores"
],
[
"Contrato de la instancia",
"Se retira tras una eliminación correcta",
"Se elimina tras una eliminación correcta",
"Ya no tiene ningún CT asociado"
],
[
@@ -190,10 +335,20 @@
"Se libera con la pila",
"Ya no le quedan miembros que conectar"
],
[
"Red que comparte la suite Arr",
"Se libera con la última aplicación de la suite",
"Cada aplicación de la suite es independiente y se elimina por separado"
],
[
"Un único miembro de una pila",
"No se elimina por separado",
"Se elimina la aplicación completa, para no dejar ninguna pila incompleta"
],
[
"Un contenedor en otro nodo del clúster",
"No se elimina",
"Su registro y sus archivos del host están en el nodo donde se instaló: se elimina allí, después de migrarlo de vuelta"
]
]
}
@@ -227,7 +382,53 @@
"title": "Registro y limpieza",
"blocks": [
{
"p": "Los contratos registrados se comparan con los CT reales. Un contrato solo queda huérfano cuando su VMID ya no existe o ya no lleva la identidad de instancia esperada. La limpieza no elimina volúmenes ni datos externos por deducción."
"p": "Al abrir <strong>Gestionar aplicaciones OCI instaladas</strong>, primero se elimina lo que quede de contenedores que no existen en ningún nodo del clúster: el registro guardado de un contenedor borrado desde la interfaz de Proxmox, que se conserva como historial, y los archivos del host que dejara una eliminación. Un registro con una operación a medias se conserva, porque su backup puede hacer falta. Los volúmenes y los directorios del host nunca se eliminan por deducción."
}
]
},
{
"id": "cluster",
"title": "En un clúster: migración y alta disponibilidad",
"intro": "Un contenedor OCI es un LXC normal de Proxmox: se puede migrar o gestionar con HA como cualquier otro, con los mismos límites. Lo que necesita para arrancar se guarda donde lo encuentra cualquier nodo del clúster.",
"blocks": [
{
"table": {
"headers": [
"Parte",
"En otro nodo del clúster"
],
"rows": [
[
"Log de consola",
"El contenedor crea <code>/var/log/proxmenux/oci</code> antes de arrancar, en el nodo que lo ejecute. Cada nodo guarda el log de los arranques que ejecutó."
],
[
"Sysctl de red y monitor del host",
"Se guardan en <code>/etc/pve/proxmenux</code>, que comparten todos los nodos del clúster, así que un contenedor migrado los encuentra."
],
[
"Discos del contenedor",
"Proxmox los mueve con el contenedor. HA necesita que estén en un almacenamiento compartido."
],
[
"Directorios del host y dispositivos",
"Las mismas reglas que cualquier LXC: un directorio del host tiene que existir en el nodo de destino y estar marcado como compartido, y una GPU, un Coral o una NPU tienen que estar presentes allí."
],
[
"Registro de ProxMenux",
"El contrato se queda en el nodo donde se instaló la aplicación, y todos los nodos leen la copia guardada en <code>/etc/pve/priv/proxmenux/oci</code>. Una aplicación de un solo contenedor que migra queda registrada en el nodo al que llega al abrir <strong>Gestionar aplicaciones OCI instaladas</strong> o al lanzar una operación sobre ella. Una aplicación de varios contenedores se ofrece para recuperarla allí, porque su red privada tiene que crearse en ese nodo."
]
]
}
},
{
"calloutWarning": {
"title": "No son para alta disponibilidad",
"body": "Una aplicación de varios contenedores comunica a sus miembros por un bridge privado del nodo donde se instaló, así que sus miembros se quedan en ese nodo. Un monitor del host, como Glances en modo host o Netdata, vigila el nodo en el que corre; moverlo haría que vigilara otro nodo."
}
},
{
"p": "Un backup restaurado fuera del clúster recibe los archivos de <code>/etc/pve/proxmenux</code> que usa el contenedor al recuperar la aplicación."
}
]
},
@@ -108,11 +108,15 @@
],
[
"Recrear",
"El editor de la recreación (recursos, red, rutas y GPU). No se ofrece en una aplicación multicontenedor."
"El editor de la recreación (recursos, red, rutas y GPU). En una aplicación multicontenedor añade o elimina las rutas y los dispositivos extra del contenedor de la aplicación y, en Immich, cambia qué ejecuta el reconocimiento."
],
[
"Recuperar",
"Sustituye a Actualizar cuando una operación sobre el contenedor quedó interrumpida, y abre su recuperación."
],
[
"Recuperar (contenedor restaurado)",
"Aparece en un contenedor restaurado desde un backup que no tiene contrato en este host, bajo <strong>Aplicación OCI restaurada</strong>. Abre la <lifecycleLink>recuperación</lifecycleLink> en el terminal del Monitor; Actualizar y Recrear aparecen cuando queda registrado."
]
]
}
+10 -1
View File
@@ -171,7 +171,7 @@
{
"id": "manage",
"title": "Gestión de una pila instalada",
"intro": "En <strong>Gestionar aplicaciones OCI instaladas</strong>, cualquier miembro lleva a la pila completa. El menú de una pila ofrece <strong>Actualizar cada contenedor de la aplicación</strong> y <strong>Eliminar: la aplicación y sus contenedores</strong>.",
"intro": "En <strong>Gestionar aplicaciones OCI instaladas</strong>, cualquier miembro lleva a la pila completa. El menú de una pila ofrece <strong>Actualizar cada contenedor de la aplicación</strong>, <strong>Recrear: añadir o eliminar rutas y dispositivos extra</strong> y <strong>Eliminar: la aplicación y sus contenedores</strong>.",
"blocks": [
{
"steps": {
@@ -185,6 +185,15 @@
]
}
},
{
"table": {
"headers": ["Recrear", "Qué cambia", "Qué se conserva"],
"rows": [
["Rutas y dispositivos extra", "Se añaden al contenedor de la aplicación o se eliminan de él, y el contenedor se reinicia. No se reconstruye nada", "Los datos de la aplicación, su base de datos y los demás contenedores"],
["Qué ejecuta el reconocimiento (Immich)", "El contenedor Machine learning toma la imagen y los dispositivos de la CPU o de una GPU del host. Toda la aplicación se detiene y se actualiza, como en una actualización", "La caché de modelos, la biblioteca y la base de datos. Si algo falla, se restauran los contenedores y la opción anteriores"]
]
}
},
{
"calloutWarning": {
"title": "Una pila sin reproducción coordinada no se actualiza",