feat(oci): GPU selection per host and per image, one notification per update, and App tab for stack containers

- Immich asks what runs its video and its recognition in one menu, in both
  modes, and gives the GPU to the server and to Machine learning; AMD uses ROCm
- Frigate, Ollama, llama.cpp, Faster Whisper and Piper take the image built
  for the chosen GPU
- The acceleration menu offers only what the host can run
- An update or a recreation sends one notification with its result instead of
  the stop, backup and start of each container
- A private bridge with nothing connected is not reported as down
- Secondary containers of a stack appear in the App tab with their version and
  logo; Secure Gateway shows the same update state in both views
- A mistyped value in the wizard asks the same question again
This commit is contained in:
MacRimi
2026-10-02 21:45:38 +02:00
parent 20ee21c08f
commit 9b5cefb81a
55 changed files with 2294 additions and 113 deletions
@@ -0,0 +1,79 @@
"""An application offers the acceleration profiles the host can run."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import DEFAULT_MODE, build_deployment
from test_advanced_flow_order import RecordingUI, storages
NODE = "/dev/dri/renderD128"
class OptionsUI(RecordingUI):
def __init__(self, answers=None):
super().__init__(answers)
self.options, self.messages = {}, []
def choose(self, text, options, default=None):
self.options[text] = [tag for tag, _ in options]
return super().choose(text, options, default)
def message(self, text, title=None):
self.messages.append(text)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
class HostProfileTests(unittest.TestCase):
catalog = Catalog(ROOT)
def offered(self, app, gpus, kfd=False, answer=None):
template = self.catalog.compose(app)
prompt = template["proxmox"]["installer_profile"]["hardware_acceleration"]["prompt"]
ui = OptionsUI({prompt: answer} if answer else None)
with patch("proxmenux_oci.installer.host.gpus", return_value=gpus), \
patch("pathlib.Path.is_char_device", return_value=kfd):
plan = build_deployment(template, ui, DEFAULT_MODE)
return ui.options.get(prompt), ui, plan
def test_an_amd_host_is_not_offered_nvidia(self, *_):
amd = {"intel": [], "amd": [NODE], "nvidia": False}
self.assertEqual(self.offered("frigate", amd, kfd=True)[0], ["none", "vaapi", "rocm"])
self.assertEqual(self.offered("ollama", amd, kfd=True)[0], ["cpu", "rocm"])
self.assertEqual(self.offered("llamacpp", amd, kfd=True)[0], ["cpu", "rocm"])
def test_rocm_is_not_offered_without_its_compute_device(self, *_):
amd = {"intel": [], "amd": [NODE], "nvidia": False}
self.assertEqual(self.offered("frigate", amd, kfd=False)[0], ["none", "vaapi"])
def test_an_intel_and_nvidia_host_is_not_offered_amd(self, *_):
both = {"intel": [NODE], "amd": [], "nvidia": True}
self.assertEqual(self.offered("frigate", both)[0], ["none", "vaapi", "nvidia"])
self.assertEqual(self.offered("llamacpp", both)[0], ["cpu", "nvidia", "intel"])
def test_a_host_without_gpu_is_told_and_not_asked(self, *_):
nothing = {"intel": [], "amd": [], "nvidia": False}
for app in ("faster-whisper", "ollama", "frigate"):
options, ui, plan = self.offered(app, nothing)
self.assertIsNone(options, app)
self.assertTrue(any("No usable GPU" in message for message in ui.messages), app)
self.assertEqual(plan["devices"], [], app)
def test_the_render_node_proposed_belongs_to_the_gpu_of_the_profile(self, *_):
# The first render node of this host is the NVIDIA one; Intel's is the second.
gpus = {"intel": ["/dev/dri/renderD129"], "amd": [], "nvidia": True}
_, _, plan = self.offered("llamacpp", gpus, answer="intel")
self.assertEqual([device["host_path"] for device in plan["devices"]], ["/dev/dri/renderD129"])
if __name__ == "__main__":
unittest.main()
+59
View File
@@ -0,0 +1,59 @@
"""The AI applications whose image depends on the GPU take the image and the
devices of the profile that is chosen."""
from pathlib import Path
import json
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import DEFAULT_MODE, build_deployment
from test_advanced_flow_order import RecordingUI, storages
RENDER, KFD = "/dev/dri/renderD128", "/dev/kfd"
EXPECTED = {
"ollama": {"cpu": (":latest", []), "nvidia": (":latest", ["nvidia-runtime"]), "rocm": (":rocm", [RENDER, KFD])},
"llamacpp": {"cpu": (":server", []), "nvidia": (":server-cuda", ["nvidia-runtime"]),
"rocm": (":server-rocm", [RENDER, KFD]), "intel": (":server-intel", [RENDER])},
"faster-whisper": {"cpu": (":latest", []), "nvidia": (":gpu", ["nvidia-runtime"])},
"piper": {"cpu": (":latest", []), "nvidia": (":gpu", ["nvidia-runtime"])},
}
SOURCES = {"ollama": "overlays", "llamacpp": "curated", "faster-whisper": "overlays", "piper": "overlays"}
# Every profile is offered: what the host has is checked in its own test.
@patch("proxmenux_oci.installer.host.gpus", return_value={"intel": ["/dev/dri/renderD128"], "amd": ["/dev/dri/renderD128"], "nvidia": True})
@patch("proxmenux_oci.installer._profile_usable", return_value=True)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
class AiGpuProfileTests(unittest.TestCase):
catalog = Catalog(ROOT)
def test_each_profile_installs_its_image_with_its_devices(self, *_):
for app, profiles in EXPECTED.items():
hardware = self.catalog.compose(app)["proxmox"]["installer_profile"]["hardware_acceleration"]
self.assertEqual([profile["id"] for profile in hardware["profiles"]], list(profiles), app)
self.assertEqual(hardware["default"], "cpu", app)
for profile, (tag, devices) in profiles.items():
template = self.catalog.compose(app)
plan = build_deployment(template, RecordingUI({hardware["prompt"]: profile}), DEFAULT_MODE)
self.assertTrue(template["container_contract"]["image"]["reference"].endswith(tag), (app, profile))
self.assertEqual([device.get("host_path") or device["kind"] for device in plan["devices"]],
devices, (app, profile))
def test_the_shipped_copy_matches_its_source(self, *_):
read = lambda place, app: json.loads((ROOT / f"catalog/{place}/{app}.json").read_text())[
"proxmox"]["installer_profile"]["hardware_acceleration"]
for app, source in SOURCES.items():
self.assertEqual(read(source, app), read("apps", app), app)
if __name__ == "__main__":
unittest.main()
+3 -1
View File
@@ -35,6 +35,8 @@ def storages_used(plan):
return used
# The GPUs of the host the tests run on are not part of what they check.
@patch("proxmenux_oci.installer.host.gpus", return_value={"intel": ["/dev/dri/renderD128"], "amd": [], "nvidia": False})
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@@ -61,7 +63,7 @@ class DefaultInstallEssentialsTests(unittest.TestCase):
"Nextcloud volume size in GB", ADDRESS, "Start the stack with Proxmox",
"Start when finished"],
"immich": [STORAGE, "Where to store the Immich library", "Library size in GB", ADDRESS,
"Start the stack with Proxmox", "Start when finished"],
"Hardware acceleration for Immich", "Start the stack with Proxmox", "Start when finished"],
"tandoor": [STORAGE, "Where to store the recipe images and files", "Files volume size in GB", ADDRESS,
"Start the stack with Proxmox"],
"paperless-ngx": [STORAGE, "Documents volume size in GB", "Where to store the consume and export folders",
+55
View File
@@ -0,0 +1,55 @@
"""Frigate's AMD profile takes the image built for ROCm and the two devices
it needs, and says what is left for the user to set."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import DEFAULT_MODE, build_deployment
from test_advanced_flow_order import RecordingUI, storages
PROMPT = "Hardware acceleration for Frigate"
# Every profile is offered: what the host has is checked in its own test.
@patch("proxmenux_oci.installer.host.gpus", return_value={"intel": ["/dev/dri/renderD128"], "amd": ["/dev/dri/renderD128"], "nvidia": True})
@patch("proxmenux_oci.installer._profile_usable", return_value=True)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
class FrigateAmdProfileTests(unittest.TestCase):
def build(self, profile):
template = Catalog(ROOT).compose("frigate")
plan = build_deployment(template, RecordingUI({PROMPT: profile}), DEFAULT_MODE)
return template, plan
def test_the_amd_profile_uses_the_rocm_image_with_both_devices(self, *_):
template, plan = self.build("rocm")
self.assertEqual(template["container_contract"]["image"]["reference"],
"ghcr.io/blakeblackshear/frigate:stable-rocm")
self.assertEqual([device["host_path"] for device in plan["devices"]], ["/dev/dri/renderD128", "/dev/kfd"])
self.assertEqual(plan["devices"][0]["drm_vendor_ids"], ["0x1002"])
self.assertTrue(any("type: onnx" in note for note in plan["completion_notes"]))
def test_the_other_profiles_keep_their_image(self, *_):
for profile, tag in (("none", "stable"), ("vaapi", "stable"), ("nvidia", "stable-tensorrt")):
template, plan = self.build(profile)
self.assertTrue(template["container_contract"]["image"]["reference"].endswith(":" + tag), profile)
self.assertFalse(plan.get("completion_notes"), profile)
def test_the_shipped_copy_matches_the_curated_profile(self, *_):
import json
read = lambda name: json.loads((ROOT / f"catalog/{name}/frigate.json").read_text())[
"proxmox"]["installer_profile"]["hardware_acceleration"]
self.assertEqual(read("curated"), read("apps"))
if __name__ == "__main__":
unittest.main()
+129
View File
@@ -0,0 +1,129 @@
"""Immich asks in one menu, in both installation modes, what runs its video
transcoding and its recognition: the CPU, or each usable GPU of the host for
both or for only one of them."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import ADVANCED_MODE, DEFAULT_MODE, build_deployment
from test_advanced_flow_order import RecordingUI, addresses, storages
PROMPT = "Hardware acceleration for Immich"
NODE = "/dev/dri/renderD128"
INTEL = {"intel": [NODE], "amd": [], "nvidia": False}
AMD = {"intel": [], "amd": [NODE], "nvidia": False}
BOTH = {"intel": [NODE], "amd": [], "nvidia": True}
NOTHING = {"intel": [], "amd": [], "nvidia": False}
class OptionsUI(RecordingUI):
def __init__(self, answers=None):
super().__init__(answers)
self.options = {}
self.defaults = {}
self.messages = []
def choose(self, text, options, default=None):
self.options[text] = [tag for tag, _ in options]
self.defaults[text] = default
return super().choose(text, options, default)
def message(self, text, title=None):
self.messages.append(text)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
@patch("proxmenux_oci.installer.host.rocm_blocker", return_value=None)
@patch("proxmenux_oci.installer.access.ask_addresses", side_effect=addresses)
class ImmichAccelerationTests(unittest.TestCase):
template = Catalog(ROOT).compose("immich")
def plan(self, gpus, mode, answer=None):
ui = OptionsUI({PROMPT: answer} if answer else None)
with patch("proxmenux_oci.installer.host.gpus", return_value=gpus):
plan = build_deployment(self.template, ui, mode)
return ui, (plan["video_transcoding"]["acceleration"], plan["machine_learning"]["acceleration"]), plan
def test_the_menu_is_asked_in_both_modes_with_what_the_host_has(self, *_):
for mode in (DEFAULT_MODE, ADVANCED_MODE):
ui, _, _ = self.plan(INTEL, mode)
self.assertEqual(ui.options[PROMPT], ["cpu", "intel", "intel-video", "intel-ml"], mode)
ui, _, _ = self.plan(BOTH, DEFAULT_MODE)
self.assertEqual(ui.options[PROMPT], ["cpu", "intel", "intel-video", "intel-ml",
"nvidia", "nvidia-video", "nvidia-ml", "intel+nvidia"])
def test_the_first_gpu_is_proposed_whole(self, *_):
ui, result, _ = self.plan(BOTH, DEFAULT_MODE)
self.assertEqual(ui.defaults[PROMPT], "intel")
self.assertEqual(result, ("vaapi", "openvino"))
def test_every_option_gives_the_gpu_to_what_it_names(self, *_):
expected = {"cpu": ("cpu", "cpu"), "intel": ("vaapi", "openvino"), "intel-video": ("vaapi", "cpu"),
"intel-ml": ("cpu", "openvino"), "nvidia": ("nvenc", "cuda"), "nvidia-video": ("nvenc", "cpu"),
"nvidia-ml": ("cpu", "cuda"), "intel+nvidia": ("vaapi", "cuda")}
for answer, result in expected.items():
self.assertEqual(self.plan(BOTH, DEFAULT_MODE, answer)[1], result, answer)
for answer, result in {"amd": ("vaapi", "rocm"), "amd-video": ("vaapi", "cpu"),
"amd-ml": ("cpu", "rocm")}.items():
self.assertEqual(self.plan(AMD, DEFAULT_MODE, answer)[1], result, answer)
def test_recognition_alone_still_gets_its_render_device(self, *_):
for gpus, answer in ((INTEL, "intel-ml"), (AMD, "amd-ml")):
_, _, plan = self.plan(gpus, DEFAULT_MODE, answer)
self.assertEqual(plan["machine_learning"]["render_device"], NODE, answer)
self.assertIsNone(plan["video_transcoding"]["render_device"], answer)
def test_a_host_without_usable_gpu_says_so_and_installs_on_the_cpu(self, *_):
for mode in (DEFAULT_MODE, ADVANCED_MODE):
ui, result, _ = self.plan(NOTHING, mode)
self.assertNotIn(PROMPT, ui.options, mode)
self.assertEqual(len(ui.messages), 1, mode)
self.assertIn("No usable GPU", ui.messages[0])
self.assertEqual(result, ("cpu", "cpu"), mode)
def test_an_amd_host_that_cannot_run_rocm_says_so_and_recognises_on_the_cpu(self, *_):
for blocker, text in (("kfd", "/dev/kfd"), ("space", "40 GB")):
with patch("proxmenux_oci.installer.host.rocm_blocker", return_value=blocker):
ui, result, _ = self.plan(AMD, DEFAULT_MODE, "amd")
self.assertEqual(result, ("vaapi", "cpu"), blocker)
self.assertTrue(any(text in message and "Recognition runs on the CPU." in message
for message in ui.messages), blocker)
def test_machine_learning_gets_four_cores_and_at_least_four_gigabytes(self, *_):
_, _, plan = self.plan(BOTH, DEFAULT_MODE, "cpu")
self.assertEqual(plan["machine_learning"]["resources"]["cores"], 4)
self.assertEqual(plan["machine_learning"]["resources"]["memory_mb"], 4096)
for gpus, answer in ((BOTH, "nvidia"), (INTEL, "intel"), (AMD, "amd")):
_, _, plan = self.plan(gpus, DEFAULT_MODE, answer)
self.assertEqual(plan["machine_learning"]["resources"]["memory_mb"], 8192, answer)
def test_the_installers_give_the_gpu_to_both_containers(self, *_):
script = (ROOT / "remote/install_immich_stack.sh").read_text()
helper = (ROOT / "remote/oci_immich_ml.sh").read_text()
self.assertIn('configure_immich_nvidia "$SERVER_ID" "compute,video,utility"', script)
self.assertIn('configure_immich_nvidia "$ML_ID" "compute,utility"', helper)
self.assertIn('--dev1 "path=/dev/kfd', helper)
self.assertIn("MIGraphXExecutionProvider", helper)
self.assertIn("ML_ROOTFS_SIZE=40", helper)
self.assertIn('--rootfs "${ROOTFS_STORAGE}:${ML_ROOTFS_SIZE}"', script)
self.assertIn("'rocm')", (ROOT / "remote/oci_stack_replay.py").read_text())
self.assertIn("MIGraphXExecutionProvider", (ROOT / "remote/oci_stack_native.py").read_text())
def test_the_name_of_the_machine_learning_container_is_not_translated(self, *_):
script = (ROOT / "remote/install_immich_stack.sh").read_text()
self.assertNotIn('translate "Machine learning"', script)
self.assertNotIn('translate("Machine learning")', (ROOT / "src/proxmenux_oci/cli.py").read_text())
if __name__ == "__main__":
unittest.main()
+62
View File
@@ -0,0 +1,62 @@
"""An update or a recreation marks its containers for the Monitor and reports
its result once, whether it works or fails."""
import json
from pathlib import Path
import sys
import tempfile
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
import oci_operation_notice as notice
class OperationNoticeTests(unittest.TestCase):
def setUp(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
self.markers = Path(tmp.name)
patcher = patch.object(notice, "MARKERS", self.markers)
patcher.start()
self.addCleanup(patcher.stop)
def mark(self, vmid):
return json.loads((self.markers / str(vmid)).read_text())
def test_the_containers_are_marked_while_it_runs_and_the_result_is_sent(self):
with patch.object(notice, "notify") as notify:
with notice.operation([115, 116], "update", "Immich", 115):
self.assertIsNone(self.mark(115)["ended"])
self.assertIsNone(self.mark(116)["ended"])
notify.assert_not_called()
self.assertIsNotNone(self.mark(115)["ended"])
notify.assert_called_once_with("oci_update_completed",
{"app_name": "Immich", "vmid": 115, "containers": "CT 115, CT 116"})
def test_a_failure_is_reported_with_its_reason_and_raised(self):
with patch.object(notice, "notify") as notify:
with self.assertRaises(RuntimeError):
with notice.operation([120], "recreate", "Jellyfin"):
raise RuntimeError("the new image did not answer")
event, data = notify.call_args.args
self.assertEqual(event, "oci_recreate_failed")
self.assertEqual(data["reason"], "the new image did not answer")
self.assertIsNotNone(self.mark(120)["ended"])
def test_a_monitor_that_does_not_answer_never_stops_the_operation(self):
with patch.object(notice.urllib.request, "urlopen", side_effect=OSError("refused")):
self.assertFalse(notice.notify("oci_update_completed", {"app_name": "x"}))
with notice.operation([120], "update", "Jellyfin"):
pass
def test_both_engines_report_through_it(self):
self.assertIn("oci_operation_notice.operation", (ROOT / "remote/oci_update_current.py").read_text())
self.assertIn("oci_operation_notice.operation", (ROOT / "remote/oci_stack_native.py").read_text())
self.assertIn("oci_operation_notice.operation", (ROOT / "remote/oci_stack_modify.py").read_text())
if __name__ == "__main__":
unittest.main()
+1 -1
View File
@@ -15,7 +15,7 @@ from proxmenux_oci.cli import _deployment_summary_text
from proxmenux_oci.installer import ADVANCED_MODE, DEFAULT_MODE, InstallError, build_deployment
from test_advanced_flow_order import RecordingUI, addresses, storages
SPECIAL = {"nextcloud-stack": (2, 2048), "paperless-ngx": (2, 2048), "tandoor": (2, 2048), "immich": (4, 3072)}
SPECIAL = {"nextcloud-stack": (2, 2048), "paperless-ngx": (2, 2048), "tandoor": (2, 2048), "immich": (4, 4096)}
REMOTE = {"nextcloud-stack": "nextcloud", "paperless-ngx": "paperless", "tandoor": "tandoor", "immich": "immich"}
+64
View File
@@ -0,0 +1,64 @@
"""A value the wizard cannot accept asks that question again, with every
earlier answer kept, instead of ending the wizard."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci import cli
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import ADVANCED_MODE
from test_advanced_flow_order import RecordingUI, addresses, storages
class TypoUI(RecordingUI):
"""Types 8o for the cores the first time, and 8 the second."""
back_enabled = False
def __init__(self):
super().__init__()
self.messages = []
self.cores = iter(["8o", "8"])
def ask(self, text, default=None, required=True):
if text == "CPU cores":
self.asked.append(text)
return next(self.cores)
return super().ask(text, default, required)
def message(self, text, title=None):
self.messages.append(text)
def review(self, text, title=None, question=None, default=True):
return False
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
@patch("proxmenux_oci.installer.host.gpus", return_value={"intel": [], "amd": [], "nvidia": False})
@patch("proxmenux_oci.installer.access.ask_addresses", side_effect=addresses)
class WizardRetryTests(unittest.TestCase):
def test_a_mistyped_number_asks_the_same_question_again(self, *_):
ui = TypoUI()
cli.install_template(ui, Catalog(ROOT).compose("tandoor"), "tandoor", ADVANCED_MODE)
self.assertEqual(ui.asked.count("CPU cores"), 2)
# The answers given before the mistake are replayed, not asked again.
self.assertEqual(ui.asked.count("Stack name"), 1)
self.assertEqual(len(ui.messages), 1)
self.assertIn("The value must be a number: '8o'", ui.messages[0])
self.assertIn("Enter the value again.", ui.messages[0])
def test_an_error_before_any_answer_still_ends_the_wizard(self, *_):
from proxmenux_oci.ui import BacktrackUI
self.assertFalse(BacktrackUI(TypoUI()).retry_last(ValueError("x")))
if __name__ == "__main__":
unittest.main()