feat(oci): GPU selection per host and per image, one notification per update, and App tab for stack containers

- Immich asks what runs its video and its recognition in one menu, in both
  modes, and gives the GPU to the server and to Machine learning; AMD uses ROCm
- Frigate, Ollama, llama.cpp, Faster Whisper and Piper take the image built
  for the chosen GPU
- The acceleration menu offers only what the host can run
- An update or a recreation sends one notification with its result instead of
  the stop, backup and start of each container
- A private bridge with nothing connected is not reported as down
- Secondary containers of a stack appear in the App tab with their version and
  logo; Secure Gateway shows the same update state in both views
- A mistyped value in the wizard asks the same question again
This commit is contained in:
MacRimi
2026-10-02 21:45:38 +02:00
parent 20ee21c08f
commit 9b5cefb81a
55 changed files with 2294 additions and 113 deletions
+53 -1
View File
@@ -246,7 +246,59 @@
"owner_strategy": "mapped-root",
"only_when_mount_type": "managed-volume"
}
]
],
"hardware_acceleration": {
"prompt": "Acceleration for Faster Whisper",
"default": "cpu",
"profiles": [
{
"id": "cpu",
"label": "CPU",
"image": {
"reference": "lscr.io/linuxserver/faster-whisper:latest",
"registry": "lscr.io",
"repository": "lscr.io/linuxserver/faster-whisper",
"tag": "latest",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": []
},
{
"id": "nvidia",
"label": "NVIDIA (CUDA)",
"architectures": [
"amd64"
],
"image": {
"reference": "lscr.io/linuxserver/faster-whisper:gpu",
"registry": "lscr.io",
"repository": "lscr.io/linuxserver/faster-whisper",
"tag": "gpu",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "nvidia-runtime",
"kind": "nvidia-runtime",
"purpose": "nvidia-cuda",
"device_selection": "all-requested-by-compose"
}
],
"environment": [
{
"name": "NVIDIA_VISIBLE_DEVICES",
"value": "all"
},
{
"name": "NVIDIA_DRIVER_CAPABILITIES",
"value": "compute,utility"
}
]
}
]
}
}
},
"compatibility": {
+43
View File
@@ -884,6 +884,49 @@
}
]
},
{
"id": "rocm",
"label": "AMD (VA-API and ROCm detection)",
"architectures": [
"amd64"
],
"image": {
"reference": "ghcr.io/blakeblackshear/frigate:stable-rocm",
"registry": "ghcr.io",
"repository": "ghcr.io/blakeblackshear/frigate",
"tag": "stable-rocm",
"digest": null,
"pull_policy": "resolve-rolling-stable-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "gpu-render",
"path_prompt": "GPU render device",
"host_path_default": "/dev/dri/renderD128",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host",
"drm_vendor_ids": [
"0x1002"
]
},
{
"id": "amd-kfd",
"path_prompt": "AMD compute device",
"host_path_default": "/dev/kfd",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host"
}
],
"completion_notes": [
"Frigate uses the AMD GPU for detection once /config/config.yaml defines a detector with type: onnx."
]
},
{
"id": "nvidia",
"label": "NVIDIA (NVDEC/CUDA)",
+124 -1
View File
@@ -254,7 +254,130 @@
"working_dir": "import-from-oci-image",
"stop_signal": "import-from-oci-image"
},
"installer_profile": {},
"installer_profile": {
"hardware_acceleration": {
"prompt": "Acceleration for llama.cpp",
"default": "cpu",
"profiles": [
{
"id": "cpu",
"label": "CPU",
"image": {
"reference": "ghcr.io/ggml-org/llama.cpp:server",
"registry": "ghcr.io",
"repository": "ghcr.io/ggml-org/llama.cpp",
"tag": "server",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": []
},
{
"id": "nvidia",
"label": "NVIDIA (CUDA)",
"architectures": [
"amd64"
],
"image": {
"reference": "ghcr.io/ggml-org/llama.cpp:server-cuda",
"registry": "ghcr.io",
"repository": "ghcr.io/ggml-org/llama.cpp",
"tag": "server-cuda",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "nvidia-runtime",
"kind": "nvidia-runtime",
"purpose": "nvidia-cuda",
"device_selection": "all-requested-by-compose"
}
],
"environment": [
{
"name": "NVIDIA_VISIBLE_DEVICES",
"value": "all"
},
{
"name": "NVIDIA_DRIVER_CAPABILITIES",
"value": "compute,utility"
}
]
},
{
"id": "rocm",
"label": "AMD (ROCm)",
"architectures": [
"amd64"
],
"image": {
"reference": "ghcr.io/ggml-org/llama.cpp:server-rocm",
"registry": "ghcr.io",
"repository": "ghcr.io/ggml-org/llama.cpp",
"tag": "server-rocm",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "gpu-render",
"path_prompt": "GPU render device",
"host_path_default": "/dev/dri/renderD128",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host",
"drm_vendor_ids": [
"0x1002"
]
},
{
"id": "amd-kfd",
"path_prompt": "AMD compute device",
"host_path_default": "/dev/kfd",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host"
}
]
},
{
"id": "intel",
"label": "Intel (SYCL)",
"architectures": [
"amd64"
],
"image": {
"reference": "ghcr.io/ggml-org/llama.cpp:server-intel",
"registry": "ghcr.io",
"repository": "ghcr.io/ggml-org/llama.cpp",
"tag": "server-intel",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "gpu-render",
"path_prompt": "GPU render device",
"host_path_default": "/dev/dri/renderD128",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host",
"drm_vendor_ids": [
"0x8086"
]
}
]
}
]
}
},
"adaptations": [
{
"id": "imported-compose-source",
+59
View File
@@ -371,11 +371,30 @@
{
"id": "cpu",
"label": "CPU",
"image": {
"reference": "ollama/ollama:latest",
"registry": "docker.io",
"repository": "ollama/ollama",
"tag": "latest",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": []
},
{
"id": "nvidia",
"label": "NVIDIA (CUDA)",
"architectures": [
"amd64"
],
"image": {
"reference": "ollama/ollama:latest",
"registry": "docker.io",
"repository": "ollama/ollama",
"tag": "latest",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "nvidia-runtime",
@@ -390,6 +409,46 @@
"value": "all"
}
]
},
{
"id": "rocm",
"label": "AMD (ROCm)",
"architectures": [
"amd64"
],
"image": {
"reference": "ollama/ollama:rocm",
"registry": "docker.io",
"repository": "ollama/ollama",
"tag": "rocm",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "gpu-render",
"path_prompt": "GPU render device",
"host_path_default": "/dev/dri/renderD128",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host",
"drm_vendor_ids": [
"0x1002"
]
},
{
"id": "amd-kfd",
"path_prompt": "AMD compute device",
"host_path_default": "/dev/kfd",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host"
}
]
}
]
},
+53 -1
View File
@@ -260,7 +260,59 @@
"owner_strategy": "mapped-root",
"only_when_mount_type": "managed-volume"
}
]
],
"hardware_acceleration": {
"prompt": "Acceleration for Piper",
"default": "cpu",
"profiles": [
{
"id": "cpu",
"label": "CPU",
"image": {
"reference": "lscr.io/linuxserver/piper:latest",
"registry": "lscr.io",
"repository": "lscr.io/linuxserver/piper",
"tag": "latest",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": []
},
{
"id": "nvidia",
"label": "NVIDIA (CUDA)",
"architectures": [
"amd64"
],
"image": {
"reference": "lscr.io/linuxserver/piper:gpu",
"registry": "lscr.io",
"repository": "lscr.io/linuxserver/piper",
"tag": "gpu",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "nvidia-runtime",
"kind": "nvidia-runtime",
"purpose": "nvidia-cuda",
"device_selection": "all-requested-by-compose"
}
],
"environment": [
{
"name": "NVIDIA_VISIBLE_DEVICES",
"value": "all"
},
{
"name": "NVIDIA_DRIVER_CAPABILITIES",
"value": "compute,utility"
}
]
}
]
}
}
},
"compatibility": {
+43
View File
@@ -884,6 +884,49 @@
}
]
},
{
"id": "rocm",
"label": "AMD (VA-API and ROCm detection)",
"architectures": [
"amd64"
],
"image": {
"reference": "ghcr.io/blakeblackshear/frigate:stable-rocm",
"registry": "ghcr.io",
"repository": "ghcr.io/blakeblackshear/frigate",
"tag": "stable-rocm",
"digest": null,
"pull_policy": "resolve-rolling-stable-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "gpu-render",
"path_prompt": "GPU render device",
"host_path_default": "/dev/dri/renderD128",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host",
"drm_vendor_ids": [
"0x1002"
]
},
{
"id": "amd-kfd",
"path_prompt": "AMD compute device",
"host_path_default": "/dev/kfd",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host"
}
],
"completion_notes": [
"Frigate uses the AMD GPU for detection once /config/config.yaml defines a detector with type: onnx."
]
},
{
"id": "nvidia",
"label": "NVIDIA (NVDEC/CUDA)",
+124 -1
View File
@@ -254,7 +254,130 @@
"working_dir": "import-from-oci-image",
"stop_signal": "import-from-oci-image"
},
"installer_profile": {},
"installer_profile": {
"hardware_acceleration": {
"prompt": "Acceleration for llama.cpp",
"default": "cpu",
"profiles": [
{
"id": "cpu",
"label": "CPU",
"image": {
"reference": "ghcr.io/ggml-org/llama.cpp:server",
"registry": "ghcr.io",
"repository": "ghcr.io/ggml-org/llama.cpp",
"tag": "server",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": []
},
{
"id": "nvidia",
"label": "NVIDIA (CUDA)",
"architectures": [
"amd64"
],
"image": {
"reference": "ghcr.io/ggml-org/llama.cpp:server-cuda",
"registry": "ghcr.io",
"repository": "ghcr.io/ggml-org/llama.cpp",
"tag": "server-cuda",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "nvidia-runtime",
"kind": "nvidia-runtime",
"purpose": "nvidia-cuda",
"device_selection": "all-requested-by-compose"
}
],
"environment": [
{
"name": "NVIDIA_VISIBLE_DEVICES",
"value": "all"
},
{
"name": "NVIDIA_DRIVER_CAPABILITIES",
"value": "compute,utility"
}
]
},
{
"id": "rocm",
"label": "AMD (ROCm)",
"architectures": [
"amd64"
],
"image": {
"reference": "ghcr.io/ggml-org/llama.cpp:server-rocm",
"registry": "ghcr.io",
"repository": "ghcr.io/ggml-org/llama.cpp",
"tag": "server-rocm",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "gpu-render",
"path_prompt": "GPU render device",
"host_path_default": "/dev/dri/renderD128",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host",
"drm_vendor_ids": [
"0x1002"
]
},
{
"id": "amd-kfd",
"path_prompt": "AMD compute device",
"host_path_default": "/dev/kfd",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host"
}
]
},
{
"id": "intel",
"label": "Intel (SYCL)",
"architectures": [
"amd64"
],
"image": {
"reference": "ghcr.io/ggml-org/llama.cpp:server-intel",
"registry": "ghcr.io",
"repository": "ghcr.io/ggml-org/llama.cpp",
"tag": "server-intel",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "gpu-render",
"path_prompt": "GPU render device",
"host_path_default": "/dev/dri/renderD128",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host",
"drm_vendor_ids": [
"0x8086"
]
}
]
}
]
}
},
"adaptations": [
{
"id": "imported-compose-source",
+53 -1
View File
@@ -8,7 +8,59 @@
"owner_strategy": "mapped-root",
"only_when_mount_type": "managed-volume"
}
]
],
"hardware_acceleration": {
"prompt": "Acceleration for Faster Whisper",
"default": "cpu",
"profiles": [
{
"id": "cpu",
"label": "CPU",
"image": {
"reference": "lscr.io/linuxserver/faster-whisper:latest",
"registry": "lscr.io",
"repository": "lscr.io/linuxserver/faster-whisper",
"tag": "latest",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": []
},
{
"id": "nvidia",
"label": "NVIDIA (CUDA)",
"architectures": [
"amd64"
],
"image": {
"reference": "lscr.io/linuxserver/faster-whisper:gpu",
"registry": "lscr.io",
"repository": "lscr.io/linuxserver/faster-whisper",
"tag": "gpu",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "nvidia-runtime",
"kind": "nvidia-runtime",
"purpose": "nvidia-cuda",
"device_selection": "all-requested-by-compose"
}
],
"environment": [
{
"name": "NVIDIA_VISIBLE_DEVICES",
"value": "all"
},
{
"name": "NVIDIA_DRIVER_CAPABILITIES",
"value": "compute,utility"
}
]
}
]
}
}
}
}
+59
View File
@@ -12,11 +12,30 @@
{
"id": "cpu",
"label": "CPU",
"image": {
"reference": "ollama/ollama:latest",
"registry": "docker.io",
"repository": "ollama/ollama",
"tag": "latest",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": []
},
{
"id": "nvidia",
"label": "NVIDIA (CUDA)",
"architectures": [
"amd64"
],
"image": {
"reference": "ollama/ollama:latest",
"registry": "docker.io",
"repository": "ollama/ollama",
"tag": "latest",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "nvidia-runtime",
@@ -31,6 +50,46 @@
"value": "all"
}
]
},
{
"id": "rocm",
"label": "AMD (ROCm)",
"architectures": [
"amd64"
],
"image": {
"reference": "ollama/ollama:rocm",
"registry": "docker.io",
"repository": "ollama/ollama",
"tag": "rocm",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "gpu-render",
"path_prompt": "GPU render device",
"host_path_default": "/dev/dri/renderD128",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host",
"drm_vendor_ids": [
"0x1002"
]
},
{
"id": "amd-kfd",
"path_prompt": "AMD compute device",
"host_path_default": "/dev/kfd",
"mode": "0660",
"deny_write": false,
"gid_strategy": "host-device-gid",
"kind": "character-device",
"container_path_strategy": "same-as-host"
}
]
}
]
},
+53 -1
View File
@@ -8,7 +8,59 @@
"owner_strategy": "mapped-root",
"only_when_mount_type": "managed-volume"
}
]
],
"hardware_acceleration": {
"prompt": "Acceleration for Piper",
"default": "cpu",
"profiles": [
{
"id": "cpu",
"label": "CPU",
"image": {
"reference": "lscr.io/linuxserver/piper:latest",
"registry": "lscr.io",
"repository": "lscr.io/linuxserver/piper",
"tag": "latest",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": []
},
{
"id": "nvidia",
"label": "NVIDIA (CUDA)",
"architectures": [
"amd64"
],
"image": {
"reference": "lscr.io/linuxserver/piper:gpu",
"registry": "lscr.io",
"repository": "lscr.io/linuxserver/piper",
"tag": "gpu",
"digest": null,
"pull_policy": "resolve-selected-tag-to-architecture-digest-at-install"
},
"device_requests": [
{
"id": "nvidia-runtime",
"kind": "nvidia-runtime",
"purpose": "nvidia-cuda",
"device_selection": "all-requested-by-compose"
}
],
"environment": [
{
"name": "NVIDIA_VISIBLE_DEVICES",
"value": "all"
},
{
"name": "NVIDIA_DRIVER_CAPABILITIES",
"value": "compute,utility"
}
]
}
]
}
}
}
}
+8 -4
View File
@@ -121,7 +121,7 @@ MEDIA_SIZE=$(jq -r '.media.size_gb // empty' "$DEPLOYMENT_FILE")
MEDIA_ROOT=$(jq -r '.media.host_path // empty' "$DEPLOYMENT_FILE")
TIMEZONE=$(jqr '.timezone')
APPLICATION_CORES=$(jqr '.resources.cores // 4')
APPLICATION_MEMORY=$(jqr '.resources.memory_mb // 3072')
APPLICATION_MEMORY=$(jqr '.resources.memory_mb // 4096')
APPLICATION_SWAP=$(jqr '.resources.swap_mb // 1024')
[[ $APPLICATION_CORES =~ ^[1-9][0-9]*$ && $APPLICATION_MEMORY =~ ^[1-9][0-9]*$ && $APPLICATION_SWAP =~ ^[0-9]+$ ]] \
|| die "$(translate "Invalid resources")"
@@ -149,6 +149,8 @@ ML_IP=${ML_ADDRESS%/*}
DB_IP=${DB_ADDRESS%/*}
VALKEY_IP=${VALKEY_ADDRESS%/*}
VIDEO_ACCELERATION=$(jqr '.video_transcoding.acceleration')
[[ $VIDEO_ACCELERATION == cpu || $VIDEO_ACCELERATION == vaapi || $VIDEO_ACCELERATION == nvenc ]] \
|| die "$(translate "Video transcoding profile not implemented:") $VIDEO_ACCELERATION"
RENDER_DEVICE=$(jq -r '.video_transcoding.render_device // empty' "$DEPLOYMENT_FILE")
VAAPI_DRIVER=$(jqr '.video_transcoding.driver')
MODEL_CACHE_SIZE=$(jqr '.machine_learning.model_cache_size_gb')
@@ -417,7 +419,7 @@ set_lxc_directive "$VALKEY_ID" lxc.signal.halt SIGTERM
msg_ok "$(translate "Container created:") CT $VALKEY_ID (Valkey)"
msg_info "$(translate "Creating the container...")"
oci_create_container "$ML_ID" "$ML_ARCHIVE" --rootfs "${ROOTFS_STORAGE}:12" \
oci_create_container "$ML_ID" "$ML_ARCHIVE" --rootfs "${ROOTFS_STORAGE}:${ML_ROOTFS_SIZE}" \
--mp0 "${ROOTFS_STORAGE}:${MODEL_CACHE_SIZE},mp=/cache,backup=1" \
--hostname "${STACK_NAME}-ml" "${ML_CPU_ARGS[@]}" --memory "$ML_MEMORY" --swap "$ML_SWAP" \
--net0 "name=eth0,bridge=${FRONTEND_BRIDGE},firewall=1,host-managed=1,${ML_FRONTEND_NET},type=veth" \
@@ -442,7 +444,7 @@ rm -rf "$ML_ROOT/cache/lost+found"
chown 100000:100000 "$ML_ROOT/cache"
chmod 0755 "$ML_ROOT/cache"
oci_quiet pct unmount "$ML_ID"
msg_ok "$(translate "Container created:") CT $ML_ID ($(translate "Machine learning"))"
msg_ok "$(translate "Container created:") CT $ML_ID (Machine learning)"
SERVER_DEVICE_ARGS=()
if [[ $VIDEO_ACCELERATION == vaapi ]]; then
@@ -463,6 +465,8 @@ oci_create_container "$SERVER_ID" "$SERVER_ARCHIVE" --rootfs "${ROOTFS_STORAGE}:
created_ids+=("$SERVER_ID")
oci_apply_extra_mounts "$SERVER_ID"
oci_apply_extra_devices "$SERVER_ID"
# NVENC needs the video capability on top of what recognition uses.
[[ $VIDEO_ACCELERATION != nvenc ]] || configure_immich_nvidia "$SERVER_ID" "compute,video,utility"
oci_quiet pct mount "$SERVER_ID"
SERVER_ROOT="/var/lib/lxc/${SERVER_ID}/rootfs"
@@ -555,7 +559,7 @@ if (( START_AFTER == 1 )); then
oci_quiet pct start "$VALKEY_ID"
wait_command Valkey 30 pct exec "$VALKEY_ID" -- valkey-cli -h "$VALKEY_IP" ping
msg_ok "$(translate "Service ready:") Valkey"
ML_LABEL=$(translate "Machine learning")
ML_LABEL="Machine learning"
msg_info "$(translate "Starting the service:") $ML_LABEL"
oci_quiet pct start "$ML_ID"
wait_command "$ML_LABEL" 60 curl -fsS "http://${ML_IP}:3003/ping"
+51 -24
View File
@@ -1,9 +1,22 @@
# Immich ML prerequisites and native GPU setup; no host driver installation.
validate_immich_ml_profile() {
ML_CPU_ARGS=(--cores 2)
ML_MEMORY=2048
ML_CPU_ARGS=(--cores 4)
ML_MEMORY=4096
ML_ROOTFS_SIZE=12
case "$ML_ACCELERATION" in
cpu) ;;
rocm)
ML_RENDER_DEVICE=$(jq -er '.machine_learning.render_device' "$DEPLOYMENT_FILE")
[[ $ML_RENDER_DEVICE =~ ^/dev/dri/renderD[0-9]+$ && -c $ML_RENDER_DEVICE ]] \
|| die "$(translate "The selected AMD render device does not exist:") $ML_RENDER_DEVICE"
[[ $(cat "/sys/class/drm/${ML_RENDER_DEVICE##*/}/device/vendor") == 0x1002 ]] \
|| die "$(translate "ROCm requires the render device of an AMD GPU")"
[[ -c /dev/kfd ]] || die "$(translate "ROCm requires /dev/kfd on the host")"
ML_MEMORY=8192
# The ROCm image carries the whole AMD runtime and is several times
# larger than the others.
ML_ROOTFS_SIZE=40
;;
openvino)
ML_RENDER_DEVICE=$(jq -er '.machine_learning.render_device' "$DEPLOYMENT_FILE")
[[ $ML_RENDER_DEVICE =~ ^/dev/dri/renderD[0-9]+$ && -c $ML_RENDER_DEVICE ]] \
@@ -57,34 +70,46 @@ validate_immich_ml_profile() {
esac
}
# Gives one container of the stack the NVIDIA GPU through the dynamic hook.
# Arguments: VMID CAPABILITIES
configure_immich_nvidia() {
# Isolate the shared standalone installer's runtime context from the stack.
(
VMID=$1
CONF="/etc/pve/lxc/${VMID}.conf"
UNPRIVILEGED_FLAG=1
DEVICE='{"kind":"nvidia-runtime","runtime_mode":"dynamic"}'
NVIDIA_GID_ENV=""
DEVICE_INDEX=0
while grep -q "^dev${DEVICE_INDEX}:" "$CONF"; do DEVICE_INDEX=$((DEVICE_INDEX + 1)); done
fragment=$(mktemp)
trap 'rm -f "$fragment"' EXIT
jq -nc --arg capabilities "$2" \
'{environment:[{name:"NVIDIA_DRIVER_CAPABILITIES",value:$capabilities}]}' >"$fragment"
DEPLOYMENT_FILE=$fragment
add_character_device() {
local path=$1 mode gid
[[ -c $path && $path == /dev/nvidia* ]] || die "$(translate "Invalid NVIDIA device:") $path"
mode="0$(stat -c %a "$path")"
gid=$(stat -c %g "$path")
oci_quiet pct set "$VMID" "--dev${DEVICE_INDEX}" "path=${path},mode=${mode},gid=${gid},deny-write=0"
DEVICE_INDEX=$((DEVICE_INDEX + 1))
}
configure_nvidia_runtime
)
}
configure_immich_ml_gpu() {
case "$ML_ACCELERATION" in
openvino)
oci_quiet pct set "$ML_ID" --dev0 "path=${ML_RENDER_DEVICE},gid=$(stat -c %g "$ML_RENDER_DEVICE"),mode=0660"
;;
rocm)
oci_quiet pct set "$ML_ID" --dev0 "path=${ML_RENDER_DEVICE},gid=$(stat -c %g "$ML_RENDER_DEVICE"),mode=0660"
oci_quiet pct set "$ML_ID" --dev1 "path=/dev/kfd,gid=$(stat -c %g /dev/kfd),mode=0660"
;;
cuda)
# Isolate the shared standalone installer's runtime context from the stack.
(
VMID=$ML_ID
CONF="/etc/pve/lxc/${ML_ID}.conf"
UNPRIVILEGED_FLAG=1
DEVICE='{"kind":"nvidia-runtime","runtime_mode":"dynamic"}'
NVIDIA_GID_ENV=""
DEVICE_INDEX=0
fragment=$(mktemp)
trap 'rm -f "$fragment"' EXIT
printf '%s\n' '{"environment":[{"name":"NVIDIA_DRIVER_CAPABILITIES","value":"compute,utility"}]}' >"$fragment"
DEPLOYMENT_FILE=$fragment
add_character_device() {
local path=$1 mode gid
[[ -c $path && $path == /dev/nvidia* ]] || die "$(translate "Invalid NVIDIA device:") $path"
mode="0$(stat -c %a "$path")"
gid=$(stat -c %g "$path")
oci_quiet pct set "$VMID" "--dev${DEVICE_INDEX}" "path=${path},mode=${mode},gid=${gid},deny-write=0"
DEVICE_INDEX=$((DEVICE_INDEX + 1))
}
configure_nvidia_runtime
)
configure_immich_nvidia "$ML_ID" "compute,utility"
;;
esac
}
@@ -101,6 +126,8 @@ if profile == "openvino":
assert "OpenVINOExecutionProvider" in ort.get_available_providers()
devices = ort.capi._pybind_state.get_available_openvino_device_ids()
assert any(device.startswith("GPU") for device in devices), devices
elif profile == "rocm":
assert "MIGraphXExecutionProvider" in ort.get_available_providers(), ort.get_available_providers()
else:
assert profile == "cuda"
assert "CUDAExecutionProvider" in ort.get_available_providers()
+3
View File
@@ -40,6 +40,9 @@ def begin(root, primary, template, deployment, members, adapter):
'mounts': []}
if Path(adapter).name == 'install_immich_stack.sh' and name == 'machine-learning':
plan['machine_learning'] = copy.deepcopy(deployment.get('machine_learning', {'acceleration': 'cpu'}))
if Path(adapter).name == 'install_immich_stack.sh' and name == 'server':
# An update must know that the server transcodes with NVIDIA.
plan['video_transcoding'] = copy.deepcopy(deployment.get('video_transcoding', {'acceleration': 'cpu'}))
if Path(adapter).name in oci_stack_replay.FILES:
plan['replay_profile'] = {'adapter': Path(adapter).name, 'role': name}
if not oci_stack_replay.FILES[Path(adapter).name][name]:
+79
View File
@@ -0,0 +1,79 @@
#!/usr/bin/env python3
"""What the Monitor is told about an update or a recreation.
While the operation runs, every container it touches is stopped, backed up
and started again. Those steps belong to the operation, so the containers are
marked for the Monitor to keep their stop, start and backup notices to itself,
and one notification with the result is sent when it ends.
"""
from __future__ import annotations
import contextlib
import json
from pathlib import Path
import socket
import ssl
import time
import urllib.request
MARKERS = Path('/run/proxmenux/oci-operations')
ENDPOINTS = ('http://127.0.0.1:8008/api/internal/oci-event', 'https://127.0.0.1:8008/api/internal/oci-event')
def _write(vmid, data):
try:
MARKERS.mkdir(parents=True, exist_ok=True)
path = MARKERS / str(int(vmid))
temporary = path.with_suffix('.tmp')
temporary.write_text(json.dumps(data))
temporary.replace(path)
except OSError:
pass
def begin(vmids):
started = time.time()
for vmid in vmids:
_write(vmid, {'started': started, 'ended': None})
def end(vmids):
# The notices of the last start can arrive after the operation returned;
# the Monitor keeps the mark for a short while after `ended`.
ended = time.time()
for vmid in vmids:
_write(vmid, {'started': ended, 'ended': ended})
def notify(event, data):
"""Best effort: the operation never depends on the Monitor answering."""
payload = json.dumps({'event': event, 'hostname': socket.gethostname(), **data}).encode()
context = ssl.create_default_context()
context.check_hostname = False
context.verify_mode = ssl.CERT_NONE
for url in ENDPOINTS:
request = urllib.request.Request(url, data=payload, headers={'Content-Type': 'application/json'})
try:
with urllib.request.urlopen(request, timeout=5, context=context if url.startswith('https') else None):
return True
except (OSError, ValueError):
continue
return False
@contextlib.contextmanager
def operation(vmids, kind, application, primary=None):
"""Mark the containers for the length of an update or a recreation and
report how it ended. `kind` is 'update' or 'recreate'."""
vmids = [int(vmid) for vmid in vmids]
data = {'app_name': str(application), 'vmid': int(primary if primary is not None else vmids[0]),
'containers': ', '.join(f'CT {vmid}' for vmid in vmids)}
begin(vmids)
try:
yield
except BaseException as error:
end(vmids)
notify(f'oci_{kind}_failed', {**data, 'reason': str(error) or type(error).__name__})
raise
end(vmids)
notify(f'oci_{kind}_completed', data)
+9
View File
@@ -208,6 +208,15 @@ def modify(root, vmid, changes):
backup = instances.location(root, vmid).parent / f"config-before-recreate-{time.strftime('%Y%m%d-%H%M%S')}.conf"
backup.write_text(run('pct', 'config', str(vmid)))
backup.chmod(0o600)
import oci_operation_notice
import oci_update_current
primary_id = (record.get('stack_member') or {}).get('primary_vmid', vmid)
name = oci_update_current.application_name(instances.read(root, primary_id), primary_id)
with oci_operation_notice.operation([vmid], 'recreate', name, primary_id):
_modify(root, vmid, changes)
def _modify(root, vmid, changes):
running = is_running(vmid)
if running:
msg_info(translate('Stopping the container...'))
+9 -1
View File
@@ -373,6 +373,10 @@ class NativeAdapter:
deployment = self.records[vmid]['deployment']
if deployment.get('replay_profile') == {'adapter': 'install_immich_stack.sh', 'role': 'machine-learning'}:
acceleration = deployment.get('machine_learning', {}).get('acceleration', 'cpu')
if acceleration == 'rocm':
member_tx.run('pct', 'exec', str(vmid), '--', 'python', '-c',
'import onnxruntime as ort; '
'assert "MIGraphXExecutionProvider" in ort.get_available_providers()')
if acceleration in ('openvino', 'cuda'):
member_tx.run('pct', 'exec', str(vmid), '--', 'python', '-c',
'import sys,ctypes,onnxruntime as ort; p=sys.argv[1]; '
@@ -674,7 +678,11 @@ def run(vmid, recover=False, acknowledge_external_data=False, keep_backup=None):
msg_ok(f"{translate('Backup created in')} {keep_backup}")
else:
adapter.keep_backup = keep_backup
result = stack_tx.execute(journal, adapter, plan)
import oci_operation_notice
import oci_update_current
with oci_operation_notice.operation([member['vmid'] for member in plan['members']], 'update',
oci_update_current.application_name(primary, primary_id), primary_id):
result = stack_tx.execute(journal, adapter, plan)
msg_ok(translate('Stack update completed. Data kept.'))
return result
+18 -5
View File
@@ -68,9 +68,13 @@ def immich_record(record):
if shlex.split(runtime.get('entrypoint', '')) != expected[role]:
raise ValueError(translate('The Immich startup was modified or cannot be reproduced'))
acceleration = record['deployment'].get('machine_learning', {}).get('acceleration', 'cpu')
if role == 'machine-learning' and acceleration not in ('cpu', 'openvino', 'cuda'):
if role == 'machine-learning' and acceleration not in ('cpu', 'openvino', 'cuda', 'rocm'):
raise ValueError(translate('Immich GPU profile not validated'))
cuda = role == 'machine-learning' and acceleration == 'cuda'
video = record['deployment'].get('video_transcoding', {}).get('acceleration', 'cpu')
# NVIDIA reaches Machine learning for recognition and the server for NVENC.
capabilities = ('compute,utility' if role == 'machine-learning' and acceleration == 'cuda'
else 'compute,video,utility' if role == 'server' and video == 'nvenc' else None)
cuda = capabilities is not None
devices = [{'kind': 'nvidia-runtime', 'runtime_mode': 'dynamic'}] if cuda else []
for item in projection['native_devices']:
fields = dict(p.split('=', 1) for p in item['value'].split(',') if '=' in p)
@@ -80,17 +84,24 @@ def immich_record(record):
if oci_gpu_devices.peripheral_path(path):
devices.append(peripheral_device(fields))
continue
rocm = role == 'machine-learning' and acceleration == 'rocm'
if rocm and path == '/dev/kfd':
# The compute interface ROCm needs beside the render node.
devices.append({'kind': 'character-device', 'host_path': path, 'container_path': path,
'gid_strategy': 'host-device-gid', 'mode': fields.get('mode', '0660')})
continue
if not path or not re.fullmatch(r'/dev/dri/renderD[0-9]+', path):
raise ValueError(translate('Immich device without a validated translation'))
vendors = (['0x1002'] if rocm else ['0x8086'] if role == 'machine-learning' else ['0x8086', '0x1002'])
devices.append({'kind': 'character-device', 'host_path': path, 'container_path': path,
'gid_strategy': 'host-device-gid', 'mode': fields.get('mode', '0660'),
'drm_vendor_ids': ['0x8086'] if role == 'machine-learning' else ['0x8086', '0x1002']})
'drm_vendor_ids': vendors})
if projection['preserved_raw_runtime'] and not cuda:
raise ValueError(translate('Immich runtime without a validated translation'))
if cuda:
import oci_accelerators
candidate = {'devices': devices, 'environment': [
{'name': 'NVIDIA_DRIVER_CAPABILITIES', 'value': 'compute,utility'}]}
{'name': 'NVIDIA_DRIVER_CAPABILITIES', 'value': capabilities}]}
oci_accelerators.check(record['observed']['config'].encode(), candidate)
translated = {'compose_entrypoint': expected[role], 'command': []}
for native, target in (('lxc.init.cwd', 'working_directory'), ('lxc.signal.halt', 'halt_signal')):
@@ -101,8 +112,10 @@ def immich_record(record):
if cuda:
result['deployment']['environment'] = [e for e in result['deployment']['environment']
if e['name'] != 'NVIDIA_DRIVER_CAPABILITIES']
result['deployment']['environment'].append({'name': 'NVIDIA_DRIVER_CAPABILITIES', 'value': 'compute,utility'})
result['deployment']['environment'].append({'name': 'NVIDIA_DRIVER_CAPABILITIES', 'value': capabilities})
result['deployment']['machine_learning'] = copy.deepcopy(record['deployment'].get('machine_learning', {}))
if 'video_transcoding' in record['deployment']:
result['deployment']['video_transcoding'] = copy.deepcopy(record['deployment']['video_transcoding'])
return result
+16 -3
View File
@@ -137,6 +137,17 @@ def kept_settings(changes, deployment):
return kept
def application_name(record, vmid):
"""The name the user knows the application by, for its notifications."""
for template in ((record.get('stack') or {}).get('template') or {}, record.get('template') or {}):
title = (template.get('catalog_ui') or {}).get('title')
if isinstance(title, dict):
title = title.get('en_US') or next(iter(title.values()), '')
if isinstance(title, str) and title.strip():
return title.strip()
return f'CT {vmid}'
def update(vmid, acknowledge_external_data=False, proposal=None, keep_backup=None):
operation = 'recreate' if proposal is not None else 'update'
msg_info(translate('Checking the container before the update...') if operation == 'update'
@@ -169,9 +180,11 @@ def update(vmid, acknowledge_external_data=False, proposal=None, keep_backup=Non
kept = kept_settings(changes, desired['deployment'])
if kept:
msg_info2(f"{translate('Keeping the settings changed in Proxmox:')} {', '.join(kept)}")
transaction.apply(instances.ROOT, vmid, archive, operation, proposal=proposal,
registry_digest=digest, acknowledge_external_data=acknowledge_external_data,
keep_backup=file_storage)
import oci_operation_notice
with oci_operation_notice.operation([vmid], operation, application_name(record, vmid)):
transaction.apply(instances.ROOT, vmid, archive, operation, proposal=proposal,
registry_digest=digest, acknowledge_external_data=acknowledge_external_data,
keep_backup=file_storage)
def main():
+7 -2
View File
@@ -35,7 +35,7 @@ PUBLISHERS = {"linuxserver.io": "LinuxServer", "official": N_("Official image")}
STACK_LABELS = {
"server": N_("Server"),
"application": N_("Application"),
"machine_learning": N_("Machine learning"),
"machine_learning": "Machine learning",
"database": "PostgreSQL",
"valkey": "Valkey",
"cache": "Redis",
@@ -129,7 +129,7 @@ def _service_kind(service: dict[str, Any]) -> str:
if service.get("is_main"):
return translate("Application")
if "machine-learning" in name or "machine-learning" in image:
return translate("Machine learning")
return "Machine learning"
for key, label in SERVICE_KINDS:
if key in image:
return label
@@ -403,6 +403,11 @@ def install_template(ui, template: dict[str, Any], identifier: str, mode: str) -
break
except RestartWizard:
wizard.restart()
except (ValueError, InstallError) as error:
# A mistyped value asks that question again instead of
# sending the user back to the start of the wizard.
if not wizard.retry_last(error):
raise
finally:
wizard.close()
if not approved:
+31
View File
@@ -5,6 +5,7 @@ from __future__ import annotations
import ipaddress
import json
import re
import shutil
import subprocess
from pathlib import Path
from typing import Any
@@ -136,6 +137,36 @@ def _sysfs(path: Path) -> str:
return ""
def gpus(root: Path = Path("/")) -> dict[str, Any]:
"""The GPUs an installation can use: the render nodes of each Intel and
AMD GPU, and whether NVIDIA is usable on the host."""
vendors = {"0x8086": "intel", "0x1002": "amd"}
found: dict[str, Any] = {"intel": [], "amd": [], "nvidia": False}
for node in sorted((root / "sys/class/drm").glob("renderD*")):
vendor = vendors.get(_sysfs(node / "device/vendor").lower())
if vendor:
found[vendor].append(f"/dev/dri/{node.name}")
# NVIDIA is usable when its driver answers and the Container Toolkit is installed.
if shutil.which("nvidia-smi") and shutil.which("nvidia-container-cli"):
try:
found["nvidia"] = subprocess.run(["nvidia-smi", "-L"], capture_output=True, text=True,
timeout=15, check=False).returncode == 0
except (OSError, subprocess.TimeoutExpired):
found["nvidia"] = False
return found
def rocm_blocker(storage: str | None, needed_gb: int = 40) -> str | None:
"""Why this host cannot run recognition on an AMD GPU with ROCm, or None.
ROCm needs the compute interface of the driver and room for its image."""
if not Path("/dev/kfd").is_char_device():
return "kfd"
row = next((item for item in storages("rootdir") if item.get("storage") == storage), None)
if row is not None and gib(row.get("avail")) < needed_gb:
return "space"
return None
def usb_devices(root: Path = Path("/"), lsusb: str | None = None) -> list[dict[str, str]]:
"""USB peripherals of this node an LXC can receive, named as the Monitor
names them: a serial adapter by its tty node, any other device by its bus
+127 -46
View File
@@ -532,6 +532,10 @@ def build_deployment(
devices, selected_hardware_profile, post_start_configurations, environment = configure_acceleration(
installer_profile, environment, unprivileged, ui, mode)
devices, completion_notes = configure_detector(installer_profile, devices, ui)
# What the selected acceleration profile leaves for the user to set.
completion_notes = [*next((profile.get("completion_notes", [])
for profile in installer_profile.get("hardware_acceleration", {}).get("profiles", [])
if profile["id"] == selected_hardware_profile), []), *completion_notes]
if advanced:
from .extra_devices import ask_extra_devices
@@ -788,6 +792,84 @@ def build_rclone_mount_deployment(
}
def _ask_immich_acceleration(ui, rootfs_storage: str | None = None) -> tuple[str, str | None, str, str, str | None]:
"""What runs Immich's video transcoding (the server) and its recognition
(the Machine learning container), asked in one menu in both modes. Each
usable GPU of the host can take both, or only one of them, and the CPU is
always there."""
software = ("cpu", None, "auto", "cpu", None)
real = essential_ui(ui)
found = host.gpus()
names = {"intel": "Intel", "amd": "AMD", "nvidia": "NVIDIA"}
vendors = [vendor for vendor in names if found[vendor]]
if not vendors:
real.message(translate("No usable GPU was found on this host. Immich will be installed on the CPU."))
return software
options = [("cpu", translate("No acceleration (CPU)"))]
for vendor in vendors:
options += [(vendor, f"{names[vendor]}: {translate('video + recognition')}"),
(f"{vendor}-video", f"{names[vendor]}: {translate('video only')}"),
(f"{vendor}-ml", f"{names[vendor]}: {translate('recognition only')}")]
if found["nvidia"]:
options += [(f"{vendor}+nvidia", f"{names[vendor]}: {translate('video')} · NVIDIA: {translate('recognition')}")
for vendor in ("intel", "amd") if found[vendor]]
# The GPU matters for Immich, so the first one is proposed whole.
selected = real.choose(translate("Hardware acceleration for Immich"), options, vendors[0])
if selected is None:
raise UserCancelled(translate("Immich configuration cancelled"))
if selected == "cpu":
return software
if "+" in selected:
video_vendor, ml_vendor = selected.split("+", 1)
else:
vendor, _, use = selected.partition("-")
video_vendor = vendor if use in ("", "video") else None
ml_vendor = vendor if use in ("", "ml") else None
video_acceleration, render_device, vaapi_driver = "cpu", None, "auto"
if video_vendor == "nvidia":
video_acceleration = "nvenc"
elif video_vendor:
nodes = found[video_vendor]
render_device = nodes[0]
if len(nodes) > 1:
render_device = ui.choose(translate("VA-API render device"), [(node, node) for node in nodes], nodes[0])
drivers = ([("auto", translate("Automatic detection")), ("iHD", "Intel iHD"), ("i965", "Intel i965")]
if video_vendor == "intel" else
[("auto", translate("Automatic detection")), ("radeonsi", "AMD radeonsi")])
vaapi_driver = ui.choose(translate("VA-API driver"), drivers, "auto")
if render_device is None or vaapi_driver is None:
raise UserCancelled(translate("Immich configuration cancelled"))
video_acceleration = "vaapi"
ml_acceleration, ml_render = "cpu", None
if ml_vendor == "nvidia":
ml_acceleration = "cuda"
elif ml_vendor == "intel":
ml_acceleration, ml_render = "openvino", render_device or found["intel"][0]
elif ml_vendor == "amd":
# ROCm is checked before it is promised; when the host cannot run it,
# recognition stays on the CPU.
blocker = host.rocm_blocker(rootfs_storage)
if blocker:
reason = (translate("The AMD driver does not offer its compute interface (/dev/kfd) on this host.")
if blocker == "kfd" else
translate("The storage has less than 40 GB free for the ROCm image."))
real.message(f"{reason}\n\n{translate('Recognition runs on the CPU.')}")
else:
ml_acceleration, ml_render = "rocm", render_device or found["amd"][0]
if ml_acceleration == "rocm":
real.message(translate("Recognition on AMD uses ROCm. Its image is several times larger than the others, "
"so the first installation takes longer, and whether a GPU works with it depends "
"on its model."))
if ml_acceleration != "cpu":
ui.message(translate("GPU recognition uses 8 GB of RAM and a limit of 4 CPU equivalents. These resources "
"were tested in the lab and are not a universal minimum. Compatibility depends on the "
"GPU, the models and the kernel. NVIDIA uses the GPUs of the Toolkit inventory; Intel "
"keeps the CPU topology."))
return video_acceleration, render_device, vaapi_driver, ml_acceleration, ml_render
def _build_immich_deployment(
template: dict[str, Any],
ui: TerminalUI | DialogUI,
@@ -798,7 +880,7 @@ def _build_immich_deployment(
stack_name = ui.ask(translate("Stack name"), defaults["stack_name"])
if not re.fullmatch(r"[a-z0-9][a-z0-9-]{0,31}", stack_name):
raise InstallError(translate("The stack name only accepts lowercase letters, numbers and hyphens"))
resources = ask_application_resources(ui, 4, 3072, 1024)
resources = ask_application_resources(ui, 4, 4096, 1024)
chosen_storage = ask_default_storage(ui, defaults["rootfs_storage"])
rootfs_storage = ask_storage(ui, translate("Storage for rootfs"), "rootdir",
chosen_storage or defaults["rootfs_storage"])
@@ -833,49 +915,11 @@ def _build_immich_deployment(
extra_mounts = ask_application_extra_paths(ui, ["/data"], media_storage or rootfs_storage)
frontend_bridge = ask_bridge(ui, translate("Access bridge for Immich"), defaults["frontend_network"]["bridge"])
addresses, frontend_gateway = access.ask_addresses(
essential_ui(ui), frontend_bridge, [translate("Immich server"), translate("Immich machine learning")])
essential_ui(ui), frontend_bridge, [translate("Immich server"), "Immich Machine learning"])
server_ipv4, ml_ipv4 = addresses.values()
timezone = ui.ask(translate("Timezone"), host.timezone())
video_acceleration = ui.choose(
translate("Video transcoding acceleration"),
[("vaapi", "VA-API"), ("cpu", "CPU")],
defaults["video_transcoding"]["acceleration"],
)
if video_acceleration is None:
raise UserCancelled(translate("Immich configuration cancelled"))
render_device = None
vaapi_driver = "auto"
if video_acceleration == "vaapi":
render_device = ui.ask(
translate("VA-API render device"), defaults["video_transcoding"]["render_device"]
)
vaapi_driver = ui.choose(
translate("VA-API driver"),
[("auto", translate("Automatic detection")), ("radeonsi", "AMD radeonsi"), ("iHD", "Intel iHD"), ("i965", "Intel i965")],
defaults["video_transcoding"]["driver"],
)
if vaapi_driver is None:
raise UserCancelled(translate("Immich configuration cancelled"))
ml_acceleration = ui.choose(
translate("Acceleration for Immich smart recognition"),
[("cpu", "CPU"), ("openvino", "Intel GPU / OpenVINO"),
("cuda", translate("NVIDIA GPU / CUDA (Toolkit on the host)"))],
"cpu",
)
if ml_acceleration is None:
raise UserCancelled(translate("Immich configuration cancelled"))
if ml_acceleration not in ("cpu", "openvino", "cuda"):
raise InstallError(translate("Recognition profile not implemented"))
ml_render = None
if ml_acceleration == "openvino":
ml_render = ui.ask(translate("Intel render device for recognition"), "/dev/dri/renderD128")
if not re.fullmatch(r"/dev/dri/renderD[0-9]+", ml_render):
raise InstallError(translate("Invalid Intel render path"))
if ml_acceleration != "cpu":
ui.message(translate("GPU recognition uses 8 GB of RAM and a limit of 4 CPU equivalents. These resources "
"were tested in the lab and are not a universal minimum. Compatibility depends on the "
"GPU, the models and the kernel. NVIDIA uses the GPUs of the Toolkit inventory; Intel "
"keeps the CPU topology."))
(video_acceleration, render_device, vaapi_driver,
ml_acceleration, ml_render) = _ask_immich_acceleration(ui, rootfs_storage)
extra_devices = ask_application_extra_devices(ui, ("usb",))
return {
"deployment_kind": "immich-four-lxc-stack",
@@ -919,8 +963,8 @@ def _build_immich_deployment(
},
"machine_learning": {"acceleration": ml_acceleration, "render_device": ml_render,
"model_cache_size_gb": 8,
"resources": {"cores": 2 if ml_acceleration == "cpu" else 4,
"memory_mb": 2048 if ml_acceleration == "cpu" else 8192,
"resources": {"cores": 4,
"memory_mb": 4096 if ml_acceleration == "cpu" else 8192,
"swap_mb": 1024,
"cpu_allocation": "quota" if ml_acceleration == "openvino" else "cpuset"}},
}
@@ -1630,6 +1674,26 @@ def configure_detector(installer_profile, devices, ui, root=Path("/")):
return [*devices, device], list(chosen.get("completion_notes", []))
def _profile_usable(profile: dict[str, Any], found: dict[str, Any]) -> bool:
"""Whether the host has what an acceleration profile needs: the NVIDIA
runtime, a GPU of the vendor it is written for, or ROCm's compute device."""
vendors = {"0x8086": "intel", "0x1002": "amd"}
for request in profile.get("device_requests", []):
if request.get("kind") == "nvidia-runtime":
if not found["nvidia"]:
return False
continue
path = str(request.get("host_path_default") or "")
if path == "/dev/kfd":
if not Path(path).is_char_device():
return False
elif path.startswith("/dev/dri/"):
wanted = [vendors[item] for item in request.get("drm_vendor_ids", []) if item in vendors] or list(vendors.values())
if not any(found[vendor] for vendor in wanted):
return False
return True
def configure_acceleration(installer_profile, environment, unprivileged, ui, mode=ADVANCED_MODE):
advanced = mode != DEFAULT_MODE
devices: list[dict[str, Any]] = []
@@ -1640,14 +1704,25 @@ def configure_acceleration(installer_profile, environment, unprivileged, ui, mod
hardware = installer_profile.get("hardware_acceleration")
if hardware:
profiles = hardware.get("profiles", [])
options = [(item["id"], item["label"]) for item in profiles]
default_profile = hardware.get("default", profiles[0]["id"] if profiles else None)
# Only what this host can run is offered; the profile already in use
# stays in the list so a recreation never loses it.
found = host.gpus()
usable = [item for item in profiles
if item["id"] == default_profile or _profile_usable(item, found)]
options = [(item["id"], item["label"]) for item in usable]
asked = advanced or not installer_profile.get("selkies")
if asked and len(usable) < len(profiles) and len(usable) == 1:
# Nothing but the CPU is left: say why there is nothing to choose.
ui.message(translate("No usable GPU was found on this host. The application will be installed "
"without hardware acceleration."))
asked = False
selected_hardware_profile = (
ui.choose(
translate(hardware.get("prompt", "Hardware acceleration")),
[(tag, translate(label)) for tag, label in options],
default_profile,
) if advanced or not installer_profile.get("selkies") else default_profile
) if asked else default_profile
)
if selected_hardware_profile is None:
raise UserCancelled(translate("Acceleration configuration cancelled"))
@@ -1710,6 +1785,12 @@ def configure_acceleration(installer_profile, environment, unprivileged, ui, mod
]
devices.append(device)
else:
# The render node proposed is one of the GPU the profile is for;
# renderD128 is not always it on a host with two GPUs.
nodes = [node for vendor_id, vendor in (("0x8086", "intel"), ("0x1002", "amd"))
if vendor_id in item.get("drm_vendor_ids", []) for node in host.gpus()[vendor]]
if nodes and item["host_path_default"] not in nodes:
item = {**item, "host_path_default": nodes[0]}
if not advanced:
host_path = item["host_path_default"]
elif item.get("purpose") in ("serial", "user-selected-device"):
+3
View File
@@ -241,6 +241,9 @@ def manage_instance(project, ui, row, action=None, lifecycle_args=()):
break
except RestartWizard:
wizard.restart()
except ValueError as error:
if not wizard.retry_last(error):
raise
finally:
wizard.close()
if not approved:
+15
View File
@@ -43,6 +43,21 @@ class BacktrackUI:
def restart(self):
self.cursor = 0
def retry_last(self, error) -> bool:
"""After an answer the wizard could not accept: say why and ask that
question again, keeping every answer given before it. False when no
answer was given yet, so there is nothing to ask again."""
if not self.answers:
return False
text = str(error)
# What Python says about a number it could not read is not for the user.
if text.startswith(("invalid literal for int()", "could not convert string to float")):
text = f"{translate('The value must be a number:')} {text.rsplit(':', 1)[-1].strip()}"
self.base.message(f"{text}\n\n{translate('Enter the value again.')}")
self.answers.pop()
self.cursor = 0
return True
def _call(self, name, *args, **kwargs):
if self.cursor < len(self.answers):
saved_name, value = self.answers[self.cursor]
@@ -0,0 +1,79 @@
"""An application offers the acceleration profiles the host can run."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import DEFAULT_MODE, build_deployment
from test_advanced_flow_order import RecordingUI, storages
NODE = "/dev/dri/renderD128"
class OptionsUI(RecordingUI):
def __init__(self, answers=None):
super().__init__(answers)
self.options, self.messages = {}, []
def choose(self, text, options, default=None):
self.options[text] = [tag for tag, _ in options]
return super().choose(text, options, default)
def message(self, text, title=None):
self.messages.append(text)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
class HostProfileTests(unittest.TestCase):
catalog = Catalog(ROOT)
def offered(self, app, gpus, kfd=False, answer=None):
template = self.catalog.compose(app)
prompt = template["proxmox"]["installer_profile"]["hardware_acceleration"]["prompt"]
ui = OptionsUI({prompt: answer} if answer else None)
with patch("proxmenux_oci.installer.host.gpus", return_value=gpus), \
patch("pathlib.Path.is_char_device", return_value=kfd):
plan = build_deployment(template, ui, DEFAULT_MODE)
return ui.options.get(prompt), ui, plan
def test_an_amd_host_is_not_offered_nvidia(self, *_):
amd = {"intel": [], "amd": [NODE], "nvidia": False}
self.assertEqual(self.offered("frigate", amd, kfd=True)[0], ["none", "vaapi", "rocm"])
self.assertEqual(self.offered("ollama", amd, kfd=True)[0], ["cpu", "rocm"])
self.assertEqual(self.offered("llamacpp", amd, kfd=True)[0], ["cpu", "rocm"])
def test_rocm_is_not_offered_without_its_compute_device(self, *_):
amd = {"intel": [], "amd": [NODE], "nvidia": False}
self.assertEqual(self.offered("frigate", amd, kfd=False)[0], ["none", "vaapi"])
def test_an_intel_and_nvidia_host_is_not_offered_amd(self, *_):
both = {"intel": [NODE], "amd": [], "nvidia": True}
self.assertEqual(self.offered("frigate", both)[0], ["none", "vaapi", "nvidia"])
self.assertEqual(self.offered("llamacpp", both)[0], ["cpu", "nvidia", "intel"])
def test_a_host_without_gpu_is_told_and_not_asked(self, *_):
nothing = {"intel": [], "amd": [], "nvidia": False}
for app in ("faster-whisper", "ollama", "frigate"):
options, ui, plan = self.offered(app, nothing)
self.assertIsNone(options, app)
self.assertTrue(any("No usable GPU" in message for message in ui.messages), app)
self.assertEqual(plan["devices"], [], app)
def test_the_render_node_proposed_belongs_to_the_gpu_of_the_profile(self, *_):
# The first render node of this host is the NVIDIA one; Intel's is the second.
gpus = {"intel": ["/dev/dri/renderD129"], "amd": [], "nvidia": True}
_, _, plan = self.offered("llamacpp", gpus, answer="intel")
self.assertEqual([device["host_path"] for device in plan["devices"]], ["/dev/dri/renderD129"])
if __name__ == "__main__":
unittest.main()
+59
View File
@@ -0,0 +1,59 @@
"""The AI applications whose image depends on the GPU take the image and the
devices of the profile that is chosen."""
from pathlib import Path
import json
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import DEFAULT_MODE, build_deployment
from test_advanced_flow_order import RecordingUI, storages
RENDER, KFD = "/dev/dri/renderD128", "/dev/kfd"
EXPECTED = {
"ollama": {"cpu": (":latest", []), "nvidia": (":latest", ["nvidia-runtime"]), "rocm": (":rocm", [RENDER, KFD])},
"llamacpp": {"cpu": (":server", []), "nvidia": (":server-cuda", ["nvidia-runtime"]),
"rocm": (":server-rocm", [RENDER, KFD]), "intel": (":server-intel", [RENDER])},
"faster-whisper": {"cpu": (":latest", []), "nvidia": (":gpu", ["nvidia-runtime"])},
"piper": {"cpu": (":latest", []), "nvidia": (":gpu", ["nvidia-runtime"])},
}
SOURCES = {"ollama": "overlays", "llamacpp": "curated", "faster-whisper": "overlays", "piper": "overlays"}
# Every profile is offered: what the host has is checked in its own test.
@patch("proxmenux_oci.installer.host.gpus", return_value={"intel": ["/dev/dri/renderD128"], "amd": ["/dev/dri/renderD128"], "nvidia": True})
@patch("proxmenux_oci.installer._profile_usable", return_value=True)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
class AiGpuProfileTests(unittest.TestCase):
catalog = Catalog(ROOT)
def test_each_profile_installs_its_image_with_its_devices(self, *_):
for app, profiles in EXPECTED.items():
hardware = self.catalog.compose(app)["proxmox"]["installer_profile"]["hardware_acceleration"]
self.assertEqual([profile["id"] for profile in hardware["profiles"]], list(profiles), app)
self.assertEqual(hardware["default"], "cpu", app)
for profile, (tag, devices) in profiles.items():
template = self.catalog.compose(app)
plan = build_deployment(template, RecordingUI({hardware["prompt"]: profile}), DEFAULT_MODE)
self.assertTrue(template["container_contract"]["image"]["reference"].endswith(tag), (app, profile))
self.assertEqual([device.get("host_path") or device["kind"] for device in plan["devices"]],
devices, (app, profile))
def test_the_shipped_copy_matches_its_source(self, *_):
read = lambda place, app: json.loads((ROOT / f"catalog/{place}/{app}.json").read_text())[
"proxmox"]["installer_profile"]["hardware_acceleration"]
for app, source in SOURCES.items():
self.assertEqual(read(source, app), read("apps", app), app)
if __name__ == "__main__":
unittest.main()
+3 -1
View File
@@ -35,6 +35,8 @@ def storages_used(plan):
return used
# The GPUs of the host the tests run on are not part of what they check.
@patch("proxmenux_oci.installer.host.gpus", return_value={"intel": ["/dev/dri/renderD128"], "amd": [], "nvidia": False})
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@@ -61,7 +63,7 @@ class DefaultInstallEssentialsTests(unittest.TestCase):
"Nextcloud volume size in GB", ADDRESS, "Start the stack with Proxmox",
"Start when finished"],
"immich": [STORAGE, "Where to store the Immich library", "Library size in GB", ADDRESS,
"Start the stack with Proxmox", "Start when finished"],
"Hardware acceleration for Immich", "Start the stack with Proxmox", "Start when finished"],
"tandoor": [STORAGE, "Where to store the recipe images and files", "Files volume size in GB", ADDRESS,
"Start the stack with Proxmox"],
"paperless-ngx": [STORAGE, "Documents volume size in GB", "Where to store the consume and export folders",
+55
View File
@@ -0,0 +1,55 @@
"""Frigate's AMD profile takes the image built for ROCm and the two devices
it needs, and says what is left for the user to set."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import DEFAULT_MODE, build_deployment
from test_advanced_flow_order import RecordingUI, storages
PROMPT = "Hardware acceleration for Frigate"
# Every profile is offered: what the host has is checked in its own test.
@patch("proxmenux_oci.installer.host.gpus", return_value={"intel": ["/dev/dri/renderD128"], "amd": ["/dev/dri/renderD128"], "nvidia": True})
@patch("proxmenux_oci.installer._profile_usable", return_value=True)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
class FrigateAmdProfileTests(unittest.TestCase):
def build(self, profile):
template = Catalog(ROOT).compose("frigate")
plan = build_deployment(template, RecordingUI({PROMPT: profile}), DEFAULT_MODE)
return template, plan
def test_the_amd_profile_uses_the_rocm_image_with_both_devices(self, *_):
template, plan = self.build("rocm")
self.assertEqual(template["container_contract"]["image"]["reference"],
"ghcr.io/blakeblackshear/frigate:stable-rocm")
self.assertEqual([device["host_path"] for device in plan["devices"]], ["/dev/dri/renderD128", "/dev/kfd"])
self.assertEqual(plan["devices"][0]["drm_vendor_ids"], ["0x1002"])
self.assertTrue(any("type: onnx" in note for note in plan["completion_notes"]))
def test_the_other_profiles_keep_their_image(self, *_):
for profile, tag in (("none", "stable"), ("vaapi", "stable"), ("nvidia", "stable-tensorrt")):
template, plan = self.build(profile)
self.assertTrue(template["container_contract"]["image"]["reference"].endswith(":" + tag), profile)
self.assertFalse(plan.get("completion_notes"), profile)
def test_the_shipped_copy_matches_the_curated_profile(self, *_):
import json
read = lambda name: json.loads((ROOT / f"catalog/{name}/frigate.json").read_text())[
"proxmox"]["installer_profile"]["hardware_acceleration"]
self.assertEqual(read("curated"), read("apps"))
if __name__ == "__main__":
unittest.main()
+129
View File
@@ -0,0 +1,129 @@
"""Immich asks in one menu, in both installation modes, what runs its video
transcoding and its recognition: the CPU, or each usable GPU of the host for
both or for only one of them."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import ADVANCED_MODE, DEFAULT_MODE, build_deployment
from test_advanced_flow_order import RecordingUI, addresses, storages
PROMPT = "Hardware acceleration for Immich"
NODE = "/dev/dri/renderD128"
INTEL = {"intel": [NODE], "amd": [], "nvidia": False}
AMD = {"intel": [], "amd": [NODE], "nvidia": False}
BOTH = {"intel": [NODE], "amd": [], "nvidia": True}
NOTHING = {"intel": [], "amd": [], "nvidia": False}
class OptionsUI(RecordingUI):
def __init__(self, answers=None):
super().__init__(answers)
self.options = {}
self.defaults = {}
self.messages = []
def choose(self, text, options, default=None):
self.options[text] = [tag for tag, _ in options]
self.defaults[text] = default
return super().choose(text, options, default)
def message(self, text, title=None):
self.messages.append(text)
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
@patch("proxmenux_oci.installer.host.rocm_blocker", return_value=None)
@patch("proxmenux_oci.installer.access.ask_addresses", side_effect=addresses)
class ImmichAccelerationTests(unittest.TestCase):
template = Catalog(ROOT).compose("immich")
def plan(self, gpus, mode, answer=None):
ui = OptionsUI({PROMPT: answer} if answer else None)
with patch("proxmenux_oci.installer.host.gpus", return_value=gpus):
plan = build_deployment(self.template, ui, mode)
return ui, (plan["video_transcoding"]["acceleration"], plan["machine_learning"]["acceleration"]), plan
def test_the_menu_is_asked_in_both_modes_with_what_the_host_has(self, *_):
for mode in (DEFAULT_MODE, ADVANCED_MODE):
ui, _, _ = self.plan(INTEL, mode)
self.assertEqual(ui.options[PROMPT], ["cpu", "intel", "intel-video", "intel-ml"], mode)
ui, _, _ = self.plan(BOTH, DEFAULT_MODE)
self.assertEqual(ui.options[PROMPT], ["cpu", "intel", "intel-video", "intel-ml",
"nvidia", "nvidia-video", "nvidia-ml", "intel+nvidia"])
def test_the_first_gpu_is_proposed_whole(self, *_):
ui, result, _ = self.plan(BOTH, DEFAULT_MODE)
self.assertEqual(ui.defaults[PROMPT], "intel")
self.assertEqual(result, ("vaapi", "openvino"))
def test_every_option_gives_the_gpu_to_what_it_names(self, *_):
expected = {"cpu": ("cpu", "cpu"), "intel": ("vaapi", "openvino"), "intel-video": ("vaapi", "cpu"),
"intel-ml": ("cpu", "openvino"), "nvidia": ("nvenc", "cuda"), "nvidia-video": ("nvenc", "cpu"),
"nvidia-ml": ("cpu", "cuda"), "intel+nvidia": ("vaapi", "cuda")}
for answer, result in expected.items():
self.assertEqual(self.plan(BOTH, DEFAULT_MODE, answer)[1], result, answer)
for answer, result in {"amd": ("vaapi", "rocm"), "amd-video": ("vaapi", "cpu"),
"amd-ml": ("cpu", "rocm")}.items():
self.assertEqual(self.plan(AMD, DEFAULT_MODE, answer)[1], result, answer)
def test_recognition_alone_still_gets_its_render_device(self, *_):
for gpus, answer in ((INTEL, "intel-ml"), (AMD, "amd-ml")):
_, _, plan = self.plan(gpus, DEFAULT_MODE, answer)
self.assertEqual(plan["machine_learning"]["render_device"], NODE, answer)
self.assertIsNone(plan["video_transcoding"]["render_device"], answer)
def test_a_host_without_usable_gpu_says_so_and_installs_on_the_cpu(self, *_):
for mode in (DEFAULT_MODE, ADVANCED_MODE):
ui, result, _ = self.plan(NOTHING, mode)
self.assertNotIn(PROMPT, ui.options, mode)
self.assertEqual(len(ui.messages), 1, mode)
self.assertIn("No usable GPU", ui.messages[0])
self.assertEqual(result, ("cpu", "cpu"), mode)
def test_an_amd_host_that_cannot_run_rocm_says_so_and_recognises_on_the_cpu(self, *_):
for blocker, text in (("kfd", "/dev/kfd"), ("space", "40 GB")):
with patch("proxmenux_oci.installer.host.rocm_blocker", return_value=blocker):
ui, result, _ = self.plan(AMD, DEFAULT_MODE, "amd")
self.assertEqual(result, ("vaapi", "cpu"), blocker)
self.assertTrue(any(text in message and "Recognition runs on the CPU." in message
for message in ui.messages), blocker)
def test_machine_learning_gets_four_cores_and_at_least_four_gigabytes(self, *_):
_, _, plan = self.plan(BOTH, DEFAULT_MODE, "cpu")
self.assertEqual(plan["machine_learning"]["resources"]["cores"], 4)
self.assertEqual(plan["machine_learning"]["resources"]["memory_mb"], 4096)
for gpus, answer in ((BOTH, "nvidia"), (INTEL, "intel"), (AMD, "amd")):
_, _, plan = self.plan(gpus, DEFAULT_MODE, answer)
self.assertEqual(plan["machine_learning"]["resources"]["memory_mb"], 8192, answer)
def test_the_installers_give_the_gpu_to_both_containers(self, *_):
script = (ROOT / "remote/install_immich_stack.sh").read_text()
helper = (ROOT / "remote/oci_immich_ml.sh").read_text()
self.assertIn('configure_immich_nvidia "$SERVER_ID" "compute,video,utility"', script)
self.assertIn('configure_immich_nvidia "$ML_ID" "compute,utility"', helper)
self.assertIn('--dev1 "path=/dev/kfd', helper)
self.assertIn("MIGraphXExecutionProvider", helper)
self.assertIn("ML_ROOTFS_SIZE=40", helper)
self.assertIn('--rootfs "${ROOTFS_STORAGE}:${ML_ROOTFS_SIZE}"', script)
self.assertIn("'rocm')", (ROOT / "remote/oci_stack_replay.py").read_text())
self.assertIn("MIGraphXExecutionProvider", (ROOT / "remote/oci_stack_native.py").read_text())
def test_the_name_of_the_machine_learning_container_is_not_translated(self, *_):
script = (ROOT / "remote/install_immich_stack.sh").read_text()
self.assertNotIn('translate "Machine learning"', script)
self.assertNotIn('translate("Machine learning")', (ROOT / "src/proxmenux_oci/cli.py").read_text())
if __name__ == "__main__":
unittest.main()
+62
View File
@@ -0,0 +1,62 @@
"""An update or a recreation marks its containers for the Monitor and reports
its result once, whether it works or fails."""
import json
from pathlib import Path
import sys
import tempfile
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "remote"))
import oci_operation_notice as notice
class OperationNoticeTests(unittest.TestCase):
def setUp(self):
tmp = tempfile.TemporaryDirectory()
self.addCleanup(tmp.cleanup)
self.markers = Path(tmp.name)
patcher = patch.object(notice, "MARKERS", self.markers)
patcher.start()
self.addCleanup(patcher.stop)
def mark(self, vmid):
return json.loads((self.markers / str(vmid)).read_text())
def test_the_containers_are_marked_while_it_runs_and_the_result_is_sent(self):
with patch.object(notice, "notify") as notify:
with notice.operation([115, 116], "update", "Immich", 115):
self.assertIsNone(self.mark(115)["ended"])
self.assertIsNone(self.mark(116)["ended"])
notify.assert_not_called()
self.assertIsNotNone(self.mark(115)["ended"])
notify.assert_called_once_with("oci_update_completed",
{"app_name": "Immich", "vmid": 115, "containers": "CT 115, CT 116"})
def test_a_failure_is_reported_with_its_reason_and_raised(self):
with patch.object(notice, "notify") as notify:
with self.assertRaises(RuntimeError):
with notice.operation([120], "recreate", "Jellyfin"):
raise RuntimeError("the new image did not answer")
event, data = notify.call_args.args
self.assertEqual(event, "oci_recreate_failed")
self.assertEqual(data["reason"], "the new image did not answer")
self.assertIsNotNone(self.mark(120)["ended"])
def test_a_monitor_that_does_not_answer_never_stops_the_operation(self):
with patch.object(notice.urllib.request, "urlopen", side_effect=OSError("refused")):
self.assertFalse(notice.notify("oci_update_completed", {"app_name": "x"}))
with notice.operation([120], "update", "Jellyfin"):
pass
def test_both_engines_report_through_it(self):
self.assertIn("oci_operation_notice.operation", (ROOT / "remote/oci_update_current.py").read_text())
self.assertIn("oci_operation_notice.operation", (ROOT / "remote/oci_stack_native.py").read_text())
self.assertIn("oci_operation_notice.operation", (ROOT / "remote/oci_stack_modify.py").read_text())
if __name__ == "__main__":
unittest.main()
+1 -1
View File
@@ -15,7 +15,7 @@ from proxmenux_oci.cli import _deployment_summary_text
from proxmenux_oci.installer import ADVANCED_MODE, DEFAULT_MODE, InstallError, build_deployment
from test_advanced_flow_order import RecordingUI, addresses, storages
SPECIAL = {"nextcloud-stack": (2, 2048), "paperless-ngx": (2, 2048), "tandoor": (2, 2048), "immich": (4, 3072)}
SPECIAL = {"nextcloud-stack": (2, 2048), "paperless-ngx": (2, 2048), "tandoor": (2, 2048), "immich": (4, 4096)}
REMOTE = {"nextcloud-stack": "nextcloud", "paperless-ngx": "paperless", "tandoor": "tandoor", "immich": "immich"}
+64
View File
@@ -0,0 +1,64 @@
"""A value the wizard cannot accept asks that question again, with every
earlier answer kept, instead of ending the wizard."""
from pathlib import Path
import sys
import unittest
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT / "src"))
sys.path.insert(0, str(Path(__file__).resolve().parent))
from proxmenux_oci import cli
from proxmenux_oci.catalog import Catalog
from proxmenux_oci.installer import ADVANCED_MODE
from test_advanced_flow_order import RecordingUI, addresses, storages
class TypoUI(RecordingUI):
"""Types 8o for the cores the first time, and 8 the second."""
back_enabled = False
def __init__(self):
super().__init__()
self.messages = []
self.cores = iter(["8o", "8"])
def ask(self, text, default=None, required=True):
if text == "CPU cores":
self.asked.append(text)
return next(self.cores)
return super().ask(text, default, required)
def message(self, text, title=None):
self.messages.append(text)
def review(self, text, title=None, question=None, default=True):
return False
@patch("proxmenux_oci.i18n.language", return_value="en")
@patch("proxmenux_oci.installer.host.storages", side_effect=storages)
@patch("proxmenux_oci.installer.host.bridges", return_value=[{"iface": "vmbr0", "cidr": "192.0.2.10/24"}])
@patch("proxmenux_oci.installer.host.timezone", return_value="Europe/Madrid")
@patch("proxmenux_oci.installer.host.gpus", return_value={"intel": [], "amd": [], "nvidia": False})
@patch("proxmenux_oci.installer.access.ask_addresses", side_effect=addresses)
class WizardRetryTests(unittest.TestCase):
def test_a_mistyped_number_asks_the_same_question_again(self, *_):
ui = TypoUI()
cli.install_template(ui, Catalog(ROOT).compose("tandoor"), "tandoor", ADVANCED_MODE)
self.assertEqual(ui.asked.count("CPU cores"), 2)
# The answers given before the mistake are replayed, not asked again.
self.assertEqual(ui.asked.count("Stack name"), 1)
self.assertEqual(len(ui.messages), 1)
self.assertIn("The value must be a number: '8o'", ui.messages[0])
self.assertIn("Enter the value again.", ui.messages[0])
def test_an_error_before_any_answer_still_ends_the_wizard(self, *_):
from proxmenux_oci.ui import BacktrackUI
self.assertFalse(BacktrackUI(TypoUI()).retry_last(ValueError("x")))
if __name__ == "__main__":
unittest.main()