Update proxmenux_debug.sh

This commit is contained in:
MacRimi
2026-06-22 12:25:49 +02:00
parent 68c8c03642
commit b8038f4d41

View File

@@ -1,246 +1,330 @@
#!/usr/bin/env bash
# ==========================================================
# ProxMenux Debug Snapshot
# ProxMenux RESTORE Debug Snapshot
# ==========================================================
# Author : MacRimi
# License : GPL-3.0
# ==========================================================
# Description:
# Emits a single GitHub-pasteable text block describing the
# current host + ProxMenux state, intended for collaborator
# bug reports. The operator runs this after reproducing an
# issue (especially in the host-backup / restore flow) and
# pastes the output verbatim in the discussion thread.
# Single GitHub-pasteable block focused on RESTORE bug
# triage only. Each section corresponds to a class of
# restore failure we need to diagnose. Nothing else is
# included — no notifications data, no global guest counts,
# no system telemetry that doesn't directly explain a
# failed restore.
#
# Design rules:
# • read-only — never modifies anything on the host.
# • no network round-trip — everything is local.
# • redacts hostnames, public IPs, MAC addresses and
# credentials from the captured output before printing.
# • single self-contained script, no external deps beyond
# standard Proxmox tooling (jq optional).
# Sections (and why each one is here):
# 1. Versions — reproducibility anchor.
# 2. GPU bind state — passthrough / VFIO bugs.
# 3. Components status — what ProxMenux thinks is installed.
# 4. Latest restore plan — rollback.json + plan.env + apply-list.
# 5. Pre-restore snapshot — what the restore overwrote.
# 6. VFIO config files — current state of every file the
# switch_gpu / restore flows touch
# (presence + content).
# 7. Boot args — /etc/kernel/cmdline vs /proc/cmdline,
# proxmox-boot-tool sync state.
# 8. dmesg boot binding — order vfio-pci ↔ nvidia loaded.
# 9. Restore service — systemd unit state + on-boot log.
# 10. Cluster post-boot log — initramfs regen + component reinstall.
# 11. Guests holding the GPU — VMs whose hostpci could keep the GPU
# reserved for vfio-pci.
# 12. Recent monitor errors — fail-fast clues from the service log.
#
# Usage:
# bash /usr/local/share/proxmenux/scripts/utilities/proxmenux_debug.sh
# # or capture to a file then upload:
# bash .../proxmenux_debug.sh > /tmp/proxmenux-debug.txt
# bash /usr/local/share/proxmenux/scripts/utilities/proxmenux_debug.sh > /tmp/debug.txt
# cat /tmp/debug.txt # copy-paste the whole block in the GitHub thread
#
# Output is redacted: hostnames, IPv4 addresses, MAC addresses
# and the obvious password=, passphrase=, token= patterns are
# replaced before printing.
# ==========================================================
set -u
# ── Redaction helpers ──────────────────────────────────────
_HOSTNAME_REAL="$(hostname 2>/dev/null || echo host)"
_HOSTNAME_TOKEN="<host>"
# Replace the real hostname plus common public/PII patterns
# in any captured output. Keeps the report shareable.
_redact() {
sed -E \
-e "s/${_HOSTNAME_REAL}/${_HOSTNAME_TOKEN}/g" \
-e "s/${_HOSTNAME_REAL}/<host>/g" \
-e 's/([0-9]{1,3}\.){3}[0-9]{1,3}/<ip>/g' \
-e 's/([0-9a-fA-F]{2}:){5}[0-9a-fA-F]{2}/<mac>/g' \
-e 's/(password|passphrase|secret|token|api[_-]?key)[[:space:]]*[:=][[:space:]]*"?[^"[:space:]]+"?/\1=<redacted>/gi'
}
_header() { echo; echo "── $1 ──"; }
_kv() { printf '%-22s %s\n' "$1:" "$2"; }
_hash() {
[ -f "$1" ] && sha256sum "$1" 2>/dev/null | awk -v p="$1" '{print substr($1,1,16)"… "p}' \
|| echo "(missing) $1"
# Dump a config file inline only when it's small (≤2KB) and non-empty.
# Larger or absent files get a single status line so the section
# doesn't balloon with content we won't read anyway.
_dump_if_small() {
local f="$1"
if [ ! -e "$f" ]; then
echo " absent : $f"
return
fi
local sz
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
echo " present (${sz} bytes): $f"
if [ -s "$f" ] && [ "$sz" -lt 2048 ] 2>/dev/null; then
sed 's/^/ | /' "$f" | _redact
fi
}
# ── Banner ─────────────────────────────────────────────────
echo "============================================================"
echo " ProxMenux Debug Snapshot"
echo " ProxMenux RESTORE Debug Snapshot"
echo " Generated: $(date -Iseconds 2>/dev/null || date)"
echo "============================================================"
# ── ProxMenux + Proxmox + Hardware ─────────────────────────
_header "Versions"
# ── 1. Versions ────────────────────────────────────────────
_header "1. Versions"
PMX_VERSION="(unknown)"
if [ -f /usr/local/share/proxmenux/monitor-app/web/package.json ]; then
PMX_VERSION=$(grep -oE '"version":[[:space:]]*"[^"]*"' \
/usr/local/share/proxmenux/monitor-app/web/package.json \
| head -1 | sed -E 's/.*"version":[[:space:]]*"([^"]+)".*/\1/')
fi
PVE_VERSION="(not a Proxmox host)"
command -v pveversion >/dev/null 2>&1 && PVE_VERSION=$(pveversion 2>/dev/null | head -1)
_kv "ProxMenux" "$PMX_VERSION"
_kv "Proxmox VE" "$PVE_VERSION"
_kv "Proxmox VE" "$(command -v pveversion >/dev/null 2>&1 && pveversion 2>/dev/null | head -1)"
_kv "Kernel" "$(uname -r 2>/dev/null)"
_kv "Distro" "$(grep -E '^PRETTY_NAME=' /etc/os-release 2>/dev/null | cut -d= -f2 | tr -d '\"')"
_kv "Uptime" "$(uptime -p 2>/dev/null || uptime)"
_header "Hardware"
_kv "CPU model" "$(grep -m1 'model name' /proc/cpuinfo 2>/dev/null | cut -d: -f2 | sed 's/^ //')"
_kv "CPU cores" "$(nproc 2>/dev/null)"
_kv "RAM total" "$(grep -m1 MemTotal /proc/meminfo 2>/dev/null | awk '{printf "%.1f GB\n", $2/1024/1024}')"
_kv "Boot mode" "$([ -d /sys/firmware/efi ] && echo UEFI || echo BIOS)"
_kv "Boot mode" "$([ -d /sys/firmware/efi ] && echo UEFI || echo BIOS)"
# ── 2. GPU bind state ─────────────────────────────────────
# Most restore bugs we've hit so far revolved around the GPU
# staying bound to vfio-pci when it should have returned to the
# nvidia driver (or vice-versa). Showing the driver, override and
# IOMMU group for every PCI display device tells us at a glance
# which side of that fence the host landed on.
_header "2. GPU PCI bind state"
if command -v lspci >/dev/null 2>&1; then
_kv "NVIDIA GPU" "$(lspci -nn 2>/dev/null | grep -iE 'nvidia.*(vga|3d)' | sed -E 's/^[0-9a-f]{2}:[0-9a-f]{2}\.[0-9]+ //' | head -1 | cut -c1-90)"
_kv "AMD GPU" "$(lspci -nn 2>/dev/null | grep -iE 'amd.*(vga|3d)|advanced micro.*(vga|3d)' | sed -E 's/^[0-9a-f]{2}:[0-9a-f]{2}\.[0-9]+ //' | head -1 | cut -c1-90)"
_kv "Intel iGPU" "$(lspci -nn 2>/dev/null | grep -iE 'intel.*(vga|3d|display)' | sed -E 's/^[0-9a-f]{2}:[0-9a-f]{2}\.[0-9]+ //' | head -1 | cut -c1-90)"
fi
# Show whichever NVIDIA devices live on this host along with the
# driver they're currently bound to — the single most useful piece
# of info for any GPU passthrough / restore bug.
if lspci -d 10de: 2>/dev/null | grep -q .; then
_header "NVIDIA bind state"
while IFS= read -r bdf; do
[ -z "$bdf" ] && continue
for bdf in $(lspci -D 2>/dev/null | grep -iE 'VGA compatible controller|3D controller|Display controller' | awk '{print $1}'); do
name=$(lspci -nn -s "$bdf" 2>/dev/null \
| sed -E 's/^[0-9a-f]{4}:[0-9a-f]{2}:[0-9a-f]{2}\.[0-9]+ //' \
| sed -E 's/(VGA compatible controller|3D controller|Display controller) \[[0-9a-f]+\]:[[:space:]]*//I' \
| cut -c1-80)
drv=$(basename "$(readlink "/sys/bus/pci/devices/$bdf/driver" 2>/dev/null)" 2>/dev/null || echo "(none)")
override=$(cat "/sys/bus/pci/devices/$bdf/driver_override" 2>/dev/null || true)
echo " $bdf driver=$drv override=${override:-(unset)}"
done < <(lspci -D -d 10de: 2>/dev/null | awk '{print $1}')
echo
iommu=$(basename "$(readlink "/sys/bus/pci/devices/$bdf/iommu_group" 2>/dev/null)" 2>/dev/null || echo "-")
echo " $bdf driver=$drv override=${override:-(unset)} iommu_group=$iommu"
echo " $name"
done
fi
# nvidia-smi gets its own short summary — confirms the driver
# can actually talk to the device, not just that it's bound.
if command -v nvidia-smi >/dev/null 2>&1; then
echo " nvidia-smi:"
if command -v nvidia-smi >/dev/null 2>&1; then
nvidia-smi --query-gpu=name,driver_version,memory.total --format=csv,noheader 2>&1 | sed 's/^/ /' | head -4
else
echo " nvidia-smi not installed"
fi
nvidia-smi --query-gpu=name,driver_version --format=csv,noheader 2>&1 | sed 's/^/ /' | head -4
fi
# ── Components ─────────────────────────────────────────────
_header "ProxMenux components_status.json"
# ── 3. ProxMenux components ───────────────────────────────
_header "3. ProxMenux components_status.json"
CSF=/usr/local/share/proxmenux/components_status.json
if [ -f "$CSF" ] && command -v jq >/dev/null 2>&1; then
jq -r 'to_entries[] | " \(.key): status=\(.value.status) version=\(.value.version // "-") patched=\(.value.patched // "-")"' "$CSF"
elif [ -f "$CSF" ]; then
sed -E 's/[[:space:]]+/ /g' "$CSF" | head -40
sed 's/^/ /' "$CSF" | head -40
else
echo " (no components registered)"
fi
# ── Script integrity ───────────────────────────────────────
_header "Script SHA-256 (first 16 chars · for drift detection)"
for f in \
/usr/local/share/proxmenux/scripts/backup_restore/backup_host.sh \
/usr/local/share/proxmenux/scripts/backup_restore/lib_host_backup_common.sh \
/usr/local/share/proxmenux/scripts/backup_restore/run_scheduled_backup.sh \
/usr/local/share/proxmenux/scripts/backup_restore/apply_pending_restore.sh \
/usr/local/share/proxmenux/scripts/backup_restore/apply_cluster_postboot.sh \
/usr/local/share/proxmenux/scripts/backup_restore/restore/monitor_apply.sh \
/usr/local/share/proxmenux/scripts/backup_restore/restore/compute_rollback_plan.sh \
/usr/local/share/proxmenux/scripts/gpu_tpu/nvidia_installer.sh \
/usr/local/share/proxmenux/scripts/gpu_tpu/switch_gpu_mode.sh \
/usr/local/share/proxmenux/scripts/gpu_tpu/switch_gpu_mode_direct.sh \
/usr/local/share/proxmenux/scripts/global/utils-install-functions.sh
do
echo " $(_hash "$f")"
done
# ── Backup / restore state ────────────────────────────────
_header "Backup jobs"
JOBS_DIR=/var/lib/proxmenux/backup-jobs
if [ -d "$JOBS_DIR" ]; then
for env_f in "$JOBS_DIR"/*.env; do
[ -f "$env_f" ] || continue
name=$(basename "$env_f" .env)
backend=$(grep -E '^BACKEND=' "$env_f" 2>/dev/null | cut -d= -f2)
mode=$(grep -E '^PROFILE_MODE=' "$env_f" 2>/dev/null | cut -d= -f2)
enabled=$(grep -E '^ENABLED=' "$env_f" 2>/dev/null | cut -d= -f2)
manual=$(grep -E '^MANUAL_RUN=' "$env_f" 2>/dev/null | cut -d= -f2)
status_f="/var/log/proxmenux/backup-jobs/${name}-last.status"
result=$(grep -E '^RESULT=' "$status_f" 2>/dev/null | cut -d= -f2)
echo " $name backend=$backend profile=$mode enabled=${enabled:-?} manual=${manual:-0} last=${result:-(no runs)}"
done | sort -u
else
echo " (no jobs dir)"
fi
_header "Pending restore"
# ── 4. Latest restore plan ────────────────────────────────
# The rollback.json + plan.env + apply-on-boot.list of the
# most-recent restore tell us exactly what the operator asked
# for and what the system tried to do. Without these we can't
# tell "rollback flag never propagated" from "rollback flag
# propagated but executor skipped guests".
_header "4. Latest restore plan"
PEND=/var/lib/proxmenux/restore-pending
if [ -d "$PEND" ]; then
current=$(readlink -f "$PEND/current" 2>/dev/null)
[ -n "$current" ] && echo " current → $current"
echo " pending dirs:"
ls -1t "$PEND" 2>/dev/null | grep -vE '^current$|^completed$' | sed 's/^/ /'
echo " completed (last 3):"
ls -1t "$PEND/completed" 2>/dev/null | head -3 | sed 's/^/ /'
# Surface the latest rollback.json if present — this is the
# single most useful piece of info for restore bugs (it shows
# what the rollback plan computed before the operator clicked).
latest_completed=$(ls -t "$PEND/completed" 2>/dev/null | head -1)
if [ -n "$latest_completed" ] && [ -f "$PEND/completed/$latest_completed/rollback.json" ]; then
echo " latest rollback.json:"
sed 's/^/ /' "$PEND/completed/$latest_completed/rollback.json"
echo
if [ -f "$PEND/completed/$latest_completed/plan.env" ]; then
echo " latest plan.env:"
sed 's/^/ /' "$PEND/completed/$latest_completed/plan.env"
latest=$(ls -t "$PEND/completed" 2>/dev/null | head -1)
if [ -n "$latest" ]; then
base="$PEND/completed/$latest"
echo " pending dir: $base"
if [ -f "$base/plan.env" ]; then
echo " plan.env:"
sed 's/^/ /' "$base/plan.env"
fi
if [ -f "$base/rollback.json" ]; then
echo " rollback.json:"
sed 's/^/ /' "$base/rollback.json"
fi
if [ -f "$base/apply-on-boot.list" ]; then
count=$(wc -l < "$base/apply-on-boot.list" 2>/dev/null || echo 0)
# Drop the misleading "first 30" tag when the file already fits.
if [ "$count" -le 30 ]; then
echo " apply-on-boot.list (${count} paths):"
else
echo " apply-on-boot.list (${count} paths; first 30):"
fi
head -30 "$base/apply-on-boot.list" | sed 's/^/ /'
fi
else
echo " (no completed restores yet)"
fi
else
echo " (no pending restore dir)"
echo " (no pending-restore dir)"
fi
# ── VFIO / passthrough artefacts ──────────────────────────
_header "VFIO / passthrough artefacts on host"
# ── 5. Pre-restore preservation ───────────────────────────
# `apply_pending_restore.sh` snapshots every file it's about to
# overwrite into /var/lib/proxmenux/pre-restore/<id>-onboot/. The
# list (paths only — we don't dump contents) tells us what the
# restore touched and lets us spot files that were ON the host
# before but NOT in the backup (additive-restore orphans).
_header "5. Pre-restore preservation (paths overwritten by the restore)"
PRE=/var/lib/proxmenux/pre-restore
if [ -d "$PRE" ]; then
latest_pre=$(ls -t "$PRE" 2>/dev/null | grep -E -- '-onboot$' | head -1)
if [ -n "$latest_pre" ]; then
echo " pre-restore dir: $PRE/$latest_pre"
find "$PRE/$latest_pre" -type f 2>/dev/null \
| sed "s|^$PRE/$latest_pre||" \
| sort | sed 's/^/ /'
else
echo " (no -onboot snapshot present — apply_pending_restore never ran)"
fi
else
echo " (no pre-restore dir)"
fi
# ── 6. VFIO / passthrough config files ────────────────────
# Every file the host-mode ↔ vfio-mode switch flow touches. We
# need both PRESENCE and CONTENT here, because "the file exists
# but is empty" and "the file has the wrong vendor IDs" are two
# different bug classes (we've hit both).
_header "6. VFIO / passthrough config files (current state)"
for f in \
/etc/proxmenux/vfio-bind.bdfs \
/etc/modules-load.d/nvidia-vfio.conf \
/etc/modules-load.d/nvidia-vfio.conf.proxmenux-disabled-vfio \
/etc/udev/rules.d/10-proxmenux-vfio-bind.rules \
/etc/udev/rules.d/70-nvidia.rules \
/etc/udev/rules.d/70-nvidia.rules.proxmenux-disabled \
/etc/modprobe.d/proxmenux-nvidia-vfio-blacklist.conf \
/etc/modprobe.d/nvidia-blacklist.conf \
/etc/modprobe.d/vfio.conf \
/etc/udev/rules.d/10-proxmenux-vfio-bind.rules \
/etc/modules-load.d/nvidia-vfio.conf \
/etc/proxmenux/vfio-bind.bdfs
/etc/modprobe.d/vfio.conf
do
if [ -e "$f" ]; then
sz=$(stat -c%s "$f" 2>/dev/null || echo "?")
echo " present (${sz} bytes): $f"
else
echo " absent : $f"
fi
_dump_if_small "$f"
done
# ── Recent logs (last 30 lines each) ──────────────────────
_header "Latest cluster-postboot log (last 30 lines)"
last_pb=$(ls -t /var/log/proxmenux/proxmenux-cluster-postboot-*.log 2>/dev/null | head -1)
[ -n "$last_pb" ] && tail -30 "$last_pb" | _redact || echo " (no cluster-postboot logs)"
_header "Latest restore-onboot log (last 30 lines)"
last_ob=$(ls -t /var/log/proxmenux/proxmenux-restore-onboot-*.log 2>/dev/null | head -1)
[ -n "$last_ob" ] && tail -30 "$last_ob" | _redact || echo " (no restore-onboot logs)"
_header "Latest backup runner log (last 30 lines)"
last_bj=$(ls -t /var/log/proxmenux/backup-jobs/*.log 2>/dev/null | head -1)
[ -n "$last_bj" ] && tail -30 "$last_bj" | _redact || echo " (no backup-job logs)"
_header "proxmenux-monitor.service (last 20 lines, errors only)"
journalctl -u proxmenux-monitor.service -n 200 --no-pager 2>/dev/null \
| grep -iE 'error|exception|traceback|failed' \
| tail -20 | _redact \
|| echo " (no errors)"
# ── Notifications config + last events ─────────────────────
_header "Notifications status"
NOTIF_DB=/usr/local/share/proxmenux/monitor.db
if [ -f "$NOTIF_DB" ] && command -v sqlite3 >/dev/null 2>&1; then
echo " channels configured:"
sqlite3 "$NOTIF_DB" \
"SELECT ' '||setting_key||' = '||substr(setting_value,1,20)||'…' \
FROM user_settings \
WHERE setting_key LIKE 'notifications.%enabled' \
OR setting_key LIKE 'notifications.%digest%' \
OR setting_key LIKE 'notifications.quiet_hours%' \
OR setting_key LIKE '%.digest_last_at';" 2>/dev/null | _redact || echo " (sqlite query failed)"
echo " history (last 5):"
sqlite3 -separator ' | ' "$NOTIF_DB" \
"SELECT sent_at, channel, event_type, substr(title,1,40) FROM notification_history ORDER BY sent_at DESC LIMIT 5;" \
2>/dev/null | sed 's/^/ /' | _redact || echo " (no history)"
else
echo " (no monitor.db or sqlite3 missing)"
# ── 7. Kernel boot args ───────────────────────────────────
# /etc/kernel/cmdline is what the operator (or our scripts)
# *configured*. /proc/cmdline is what the running kernel actually
# booted with. They diverge whenever someone edits cmdline and
# forgets proxmox-boot-tool refresh — and that drift breaks every
# downstream passthrough/intel_iommu assumption.
_header "7. Kernel boot args"
echo " /etc/kernel/cmdline (configured):"
[ -f /etc/kernel/cmdline ] && sed 's/^/ /' /etc/kernel/cmdline | _redact || echo " (not present — non-systemd-boot host)"
echo " /proc/cmdline (running):"
sed 's/^/ /' /proc/cmdline 2>/dev/null | _redact
# /etc/default/grub is the OTHER place kernel args can live. On
# systemd-boot hosts `proxmox-boot-tool refresh` reads cmdline +
# GRUB_CMDLINE_LINUX_DEFAULT and concatenates them — so a drift
# here is what causes "intel_iommu=on disappears at the next
# proxmox-boot-tool refresh after restore" bugs.
if [ -f /etc/default/grub ]; then
echo " /etc/default/grub GRUB_CMDLINE_LINUX*:"
grep -E '^GRUB_CMDLINE_LINUX(_DEFAULT)?=' /etc/default/grub 2>/dev/null \
| sed 's/^/ /' | _redact || echo " (no GRUB_CMDLINE_LINUX entries)"
fi
if command -v proxmox-boot-tool >/dev/null 2>&1; then
echo " proxmox-boot-tool status:"
proxmox-boot-tool status 2>&1 | grep -vE '^(System|Re-executing)' | sed 's/^/ /' | head -10
fi
# ── Guests (count only — config-level data stays private) ──
_header "Guests on this host (count only)"
qm_count=$(qm list 2>/dev/null | awk 'NR>1' | wc -l)
pct_count=$(pct list 2>/dev/null | awk 'NR>1' | wc -l)
_kv "VMs" "$qm_count"
_kv "LXCs" "$pct_count"
# ── 8. dmesg — vfio + nvidia + GPU PCI bus ────────────────
# The early-boot trace. Tells us whether vfio-pci or nvidia
# claimed the PCI BDF FIRST. If we see `vfio-pci 0000:01:00.0`
# before `nvidia 0000:01:00.0`, that's the smoking gun for a
# restore that didn't clear the vfio bindings before reboot.
_header "8. dmesg — vfio / nvidia / GPU bus binding sequence"
dmesg 2>/dev/null | grep -iE 'vfio|nvidia|VFIO' | head -25 | _redact \
|| echo " (no matches in dmesg)"
# ── 9. Restore service status + on-boot log ───────────────
# systemd's own view of the on-boot apply (success/timeout/exit)
# plus the script's log file. The two together cover both
# "the script failed silently" and "the script ran but its log
# was truncated by a hung apply".
_header "9. proxmenux-restore-onboot.service — current boot"
journalctl -u proxmenux-restore-onboot.service -b 0 -n 20 --no-pager 2>/dev/null \
| _redact || echo " (no journal entries)"
echo
last_ob=$(ls -t /var/log/proxmenux/proxmenux-restore-onboot-*.log 2>/dev/null | head -1)
if [ -n "$last_ob" ]; then
echo " log file: $last_ob"
tail -30 "$last_ob" | sed 's/^/ /' | _redact
else
echo " (no restore-onboot log files)"
fi
# ── 10. Cluster post-boot log ─────────────────────────────
# Where update-initramfs / update-grub / component auto-reinstall
# actually run. Surface the whole tail so we see whether the
# postboot finished or stopped partway through.
_header "10. apply_cluster_postboot — last run (full log, redacted)"
last_pb=$(ls -t /var/log/proxmenux/proxmenux-cluster-postboot-*.log 2>/dev/null | head -1)
if [ -n "$last_pb" ]; then
echo " log file: $last_pb"
# Dump the WHOLE post-boot log when small enough — the operator
# almost always wants to see the header (`=== ProxMenux cluster
# post-boot apply at … ===`) AND the tail together. 50-line tail
# was eating the start of the log on every restore we've inspected.
sz=$(stat -c%s "$last_pb" 2>/dev/null || echo 0)
if [ "$sz" -lt 8192 ]; then
sed 's/^/ /' "$last_pb" | _redact
else
echo " (log > 8 KB — showing last 80 lines)"
tail -80 "$last_pb" | sed 's/^/ /' | _redact
fi
else
echo " (no cluster-postboot log files)"
fi
# ── 11. Guests holding the GPU ────────────────────────────
# If a VM with `hostpci` still exists after the restore, Proxmox
# will auto-rebind the matching PCI device to vfio-pci at the
# next boot — which makes the GPU look "stuck" in passthrough
# even though every config file looks clean. We only list IDs +
# BDFs, NOT full configs, to keep secrets out of the report.
_header "11. Guests with hostpci entries (would re-reserve the GPU)"
hits=0
if command -v qm >/dev/null 2>&1; then
for vmid in $(qm list 2>/dev/null | awk 'NR>1 {print $1}'); do
out=$(qm config "$vmid" 2>/dev/null | grep -iE '^hostpci[0-9]+:' || true)
if [ -n "$out" ]; then
echo " VM $vmid:"
echo "$out" | sed 's/^/ /'
hits=$((hits + 1))
fi
done
fi
if command -v pct >/dev/null 2>&1; then
for ctid in $(pct list 2>/dev/null | awk 'NR>1 {print $1}'); do
out=$(pct config "$ctid" 2>/dev/null | grep -iE 'mp[0-9]+:.*nvidia|dev[0-9]+:|lxc.cgroup2.devices' || true)
if [ -n "$out" ]; then
echo " CT $ctid:"
echo "$out" | sed 's/^/ /'
hits=$((hits + 1))
fi
done
fi
[ "$hits" = "0" ] && echo " (no guests hold passthrough entries)"
# ── 12. proxmenux-monitor errors (current boot) ───────────
# Any Python traceback or "Error/Exception" line that ended up
# in the service log this boot. Filtered so we don't drown the
# report in the normal startup noise.
_header "12. proxmenux-monitor.service — errors this boot"
err=$(journalctl -u proxmenux-monitor.service -b 0 --no-pager 2>/dev/null \
| grep -iE 'error|exception|traceback|failed' | tail -15)
if [ -n "$err" ]; then
echo "$err" | sed 's/^/ /' | _redact
else
echo " (no errors logged this boot)"
fi
echo
echo "============================================================"
echo " End of debug snapshot. Paste the above in the GitHub thread."
echo " End of restore debug snapshot."
echo " Paste the whole block in the GitHub discussion thread."
echo "============================================================"