318 Commits
Author SHA1 Message Date
ProxMenuxBot de8738ef32 Update helpers_cache.json 2026-08-01 02:04:35 +00:00
ProxMenuxBot c6a90361d0 Update helpers_cache.json 2026-07-31 19:11:25 +00:00
ProxMenuxBot 9c6940200e Update helpers_cache.json 2026-07-31 13:45:24 +00:00
ProxMenuxBot 7eeb9fefb1 Update helpers_cache.json 2026-07-31 08:37:16 +00:00
github-actions[bot] 42b07b5ba9 chore: update repository growth [skip ci] 2026-07-31 07:17:07 +00:00
ProxMenuxBot b9b6939eff Update helpers_cache.json 2026-07-31 02:03:03 +00:00
ProxMenuxBot aba17c82e7 Update helpers_cache.json 2026-07-30 19:12:38 +00:00
ProxMenuxBot 9a952bdb38 Update helpers_cache.json 2026-07-30 13:40:49 +00:00
ProxMenuxBot 185d94183a Update helpers_cache.json 2026-07-30 08:09:59 +00:00
github-actions[bot] e880872612 chore: update repository growth [skip ci] 2026-07-30 06:44:14 +00:00
ProxMenuxBot 8d9a913b95 Update helpers_cache.json 2026-07-29 13:50:36 +00:00
ProxMenuxBot 8976e517d3 Update helpers_cache.json 2026-07-29 08:25:14 +00:00
github-actions[bot] b9e655c8cf chore: update repository growth [skip ci] 2026-07-29 06:47:20 +00:00
ProxMenuxBot c67c24c210 Update helpers_cache.json 2026-07-28 19:11:21 +00:00
ProxMenuxBot f1dbf0ed42 Update helpers_cache.json 2026-07-28 13:45:18 +00:00
ProxMenuxBot f876b7cd20 Update helpers_cache.json 2026-07-28 08:19:22 +00:00
github-actions[bot] 5929e69793 chore: update repository growth [skip ci] 2026-07-28 06:43:13 +00:00
ProxMenuxBot 10c7b335fb Update helpers_cache.json 2026-07-27 14:14:00 +00:00
ProxMenuxBot 3d0ded5435 Update helpers_cache.json 2026-07-27 09:35:04 +00:00
github-actions[bot] 4498bd8981 chore: update repository growth [skip ci] 2026-07-27 07:55:25 +00:00
ProxMenuxBot e438e283f7 Update helpers_cache.json 2026-07-26 18:59:31 +00:00
ProxMenuxBot 3c7f4bffd3 Update helpers_cache.json 2026-07-26 13:02:46 +00:00
github-actions[bot] 04f1bbe761 chore: update repository growth [skip ci] 2026-07-26 07:06:09 +00:00
github-actions[bot] 6935932dd0 chore: update repository growth [skip ci] 2026-07-25 06:31:25 +00:00
ProxMenuxBot 082f6ec0e8 Update helpers_cache.json 2026-07-24 13:21:10 +00:00
github-actions[bot] 594e266d3f chore: update repository growth [skip ci] 2026-07-24 06:41:44 +00:00
github-actions[bot] 1f10d8908b Update AppImage release build (2026-07-23 22:09:46) 2026-07-23 22:09:46 +00:00
MacRimiandGitHub e168107a78 Merge pull request #265 from MacRimi/develop
update documentation
2026-07-24 00:04:33 +02:00
MacRimi f5aa0db9ef Merge branch 'main' into develop — resolve README modify/delete (keep develop) 2026-07-24 00:02:15 +02:00
MacRimi 03a7f58dfe Update README.md 2026-07-23 23:59:07 +02:00
MacRimiandGitHub 61c778cc80 Delete AppImage/ProxMenux-1.2.4.AppImage 2026-07-23 23:53:34 +02:00
MacRimiandGitHub fb06ed4f8a Delete AppImage/ProxMenux-Monitor.AppImage.sha256 2026-07-23 23:53:22 +02:00
MacRimiandGitHub b4a4bb3ddd Delete README.md 2026-07-23 23:53:05 +02:00
MacRimi 497934f854 update 1.2.4.1 beta 2026-07-23 23:49:31 +02:00
MacRimi 20f1372816 Update README.md 2026-07-23 23:25:21 +02:00
github-actions[bot] 34ec56324f Update AppImage beta build (2026-07-23 21:06:56) 2026-07-23 21:06:56 +00:00
ProxMenuxBot 031084decd chore(lang): auto-rebuild translation cache
Source: 91d503e
Triggered by: push
2026-07-23 20:56:37 +00:00
MacRimi 91d503ea67 new beta 1.2.4.1 2026-07-23 22:55:28 +02:00
github-actions[bot] 057136b1e0 chore: update repository growth [skip ci] 2026-07-23 14:53:22 +00:00
github-actions[bot] 1eeb44bae6 chore: update repository growth [skip ci] 2026-07-23 06:41:51 +00:00
github-actions[bot] 763ed5dcc6 chore: update repository growth [skip ci] 2026-07-22 18:20:18 +00:00
MacRimi 7bf709a679 chore: backfill repository growth history [skip ci] 2026-07-22 20:10:42 +02:00
github-actions[bot] 1e6cb58a7c chore: update repository growth [skip ci] 2026-07-22 18:05:38 +00:00
github-actions[bot] 61b45a5331 chore: update repository growth [skip ci] 2026-07-22 18:01:04 +00:00
MacRimiandGitHub b9c9a8bb13 Merge pull request #261 from MacRimi/feat/repo-growth
Replace the broken Star History embed with a repository-owned Repo Growth workflow and SVG dashboard.
2026-07-22 19:55:13 +02:00
MacRimi 06b546427b feat: replace Star History with Repo Growth 2026-07-22 19:54:05 +02:00
MacRimi 37f06b9e88 update changelog 2026-07-22 18:16:23 +02:00
MacRimiandGitHub 807edc5700 Actualiza changelog para ProxMenux v1.2.4
Añade detalles sobre mejoras y nuevas funciones en la versión 1.2.4 de ProxMenux, incluyendo el botón de actualización y optimizaciones en el flujo de restauración.
2026-07-22 18:14:10 +02:00
MacRimiandGitHub 1c0910563b Update changelog date and version details
Actualiza la fecha de la versión y detalla mejoras en el dashboard, restauración de Backups, y optimización de ZFS ARC.
2026-07-22 18:09:47 +02:00
MacRimi abf5ec7405 update changelog 2026-07-22 18:08:06 +02:00
MacRimiandGitHub d9aea2ed1d Update CHANGELOG for ProxMenux v1.2.4 release
Updated release date and added details for ProxMenux v1.2.4.
2026-07-22 18:06:37 +02:00
MacRimi 047217dc93 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-07-22 17:46:36 +02:00
MacRimi 46cc91e52b Update install_proxmenux.sh 2026-07-22 17:46:22 +02:00
github-actions[bot] d94554302f Update AppImage beta build (2026-07-22 15:44:29) 2026-07-22 15:44:29 +00:00
github-actions[bot] af5cf30e34 Update AppImage release build (2026-07-22 15:41:39) 2026-07-22 15:41:40 +00:00
MacRimiandGitHub fd3986ccfd New version 1.2.4
New version 1.2.4
2026-07-22 17:34:56 +02:00
MacRimiandGitHub 91355372b6 Delete AppImage/ProxMenux-Monitor.AppImage.sha256 2026-07-22 17:34:32 +02:00
MacRimiandGitHub 70ae8326f2 Delete AppImage/ProxMenux-1.2.4.AppImage 2026-07-22 17:34:20 +02:00
MacRimi d24d4372a7 Update es.json 2026-07-22 17:31:43 +02:00
MacRimi 2b7c498570 New version 1.2.4 2026-07-22 17:14:04 +02:00
github-actions[bot] ad90f6c99a Update AppImage beta build (2026-07-22 15:12:46) 2026-07-22 15:12:46 +00:00
github-actions[bot] 98d8badb52 Update AppImage release build (2026-07-22 15:09:50) 2026-07-22 15:09:50 +00:00
MacRimiandGitHub dda0492152 v1.2.4
This release adds two in-dashboard improvements — a one-click Proxmox update trigger from the Health Monitor and a mobile PWA install prompt — extends the Backups restore flow with atomic pmxcfs (`config.db`) snapshots and automatic ZFS data-pool import, sharpens Log2RAM behaviour on hosts running Proxmox Backup Server as a service, hardens firewall bridge sysctl tuning across VM lifecycle events, narrows the ZFS ARC optimization to its own scope, makes persistent NIC naming idempotent across reruns, rebuilds DKMS drivers automatically when a new kernel is staged, keeps the Monitor terminal session intact when a ProxMenux update is available, and reinforces five notification templates plus three Health panel checks.
2026-07-22 17:04:55 +02:00
MacRimiandGitHub 005f668d7e Delete AppImage/ProxMenux-Monitor.AppImage.sha256 2026-07-22 17:04:00 +02:00
MacRimiandGitHub 76ddc8b408 Delete AppImage/ProxMenux-1.2.3.AppImage 2026-07-22 17:03:49 +02:00
MacRimi 8645ee0744 Update 1.2.4 2026-07-22 17:02:36 +02:00
MacRimi 78c5765330 Update es.md 2026-07-22 16:46:04 +02:00
MacRimi 4f5ccc4933 update 1.2.4 2026-07-22 16:39:34 +02:00
github-actions[bot] acebb1755d Update AppImage beta build (2026-07-22 13:33:42) 2026-07-22 13:33:42 +00:00
MacRimi b461b85dec update 1.2.4 2026-07-22 15:31:33 +02:00
ProxMenuxBot 05dc2cbd6a Update helpers_cache.json 2026-07-22 13:24:54 +00:00
github-actions[bot] c8613fc864 Update AppImage beta build (2026-07-21 17:27:19) 2026-07-21 17:27:19 +00:00
MacRimi 827ccbab57 update 1.2.4 2026-07-21 19:25:08 +02:00
MacRimi c53289753d update 1.2.4 2026-07-21 19:09:59 +02:00
github-actions[bot] f2516010c7 Update AppImage beta build (2026-07-21 16:53:02) 2026-07-21 16:53:02 +00:00
ProxMenuxBot 96f88d3eb5 chore(lang): auto-rebuild translation cache
Source: 2330179
Triggered by: push
2026-07-21 16:48:24 +00:00
MacRimi 2330179036 update 1.2.4 2026-07-21 18:47:19 +02:00
github-actions[bot] fd951f134c Update AppImage beta build (2026-07-21 16:34:38) 2026-07-21 16:34:38 +00:00
ProxMenuxBot ff0f8f1133 chore(lang): auto-rebuild translation cache
Source: b2c6a6c
Triggered by: push
2026-07-21 16:30:53 +00:00
MacRimi b2c6a6c536 Update 1.2.4 2026-07-21 18:26:11 +02:00
ProxMenuxBot c81230b0f3 chore(lang): auto-rebuild translation cache
Source: cf58719
Triggered by: push
2026-07-20 17:13:04 +00:00
MacRimi cf5871981d update 1.2.4 2026-07-20 19:11:29 +02:00
ProxMenuxBot 2e0746e850 Update helpers_cache.json 2026-07-20 08:43:55 +00:00
ProxMenuxBot 1fb57c77c6 Update helpers_cache.json 2026-07-19 12:59:03 +00:00
ProxMenuxBot dcf682af76 Update helpers_cache.json 2026-07-19 08:02:36 +00:00
github-actions[bot] 9f014a9784 Update AppImage beta build (2026-07-18 19:28:41) 2026-07-18 19:28:41 +00:00
MacRimi c2442565bd Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-07-18 21:22:18 +02:00
MacRimi 6b02c13e23 update v1.2.4 2026-07-18 21:22:02 +02:00
github-actions[bot] 016681caba Update AppImage beta build (2026-07-18 19:13:04) 2026-07-18 19:13:04 +00:00
ProxMenuxBot 4630be1921 chore(lang): auto-rebuild translation cache
Source: 451f541
Triggered by: push
2026-07-18 19:10:33 +00:00
MacRimi 451f541342 new version 1.2.4 2026-07-18 21:09:40 +02:00
ProxMenuxBot ab4120a81e Update helpers_cache.json 2026-07-18 12:56:28 +00:00
ProxMenuxBot c21441a56e Update helpers_cache.json 2026-07-18 07:36:51 +00:00
ProxMenuxBot 8108c42eae Update helpers_cache.json 2026-07-17 18:56:10 +00:00
ProxMenuxBot 3c5cad6ab5 Update helpers_cache.json 2026-07-17 13:10:19 +00:00
MacRimi 52d7e20979 Update AppImage 1.2.3 2026-07-17 00:05:06 +02:00
MacRimi db79470d32 Update AppImage 1.2.3 2026-07-16 23:43:58 +02:00
MacRimi 9dc077feec update AppImage 1.2.3 2026-07-16 23:32:54 +02:00
MacRimi 1f3702b700 update AppImage 1.2.3 2026-07-16 23:19:12 +02:00
MacRimi 21c8b7a62e update AppImage 1.2.3 2026-07-16 23:12:28 +02:00
MacRimi 2770c8172b Update notification_events.py 2026-07-16 23:07:25 +02:00
github-actions[bot] f4cd661480 Update AppImage beta build (2026-07-16 21:01:14) 2026-07-16 21:01:14 +00:00
MacRimi 3f61bca07c Update AppImage 1.2.3 2026-07-16 22:58:50 +02:00
github-actions[bot] af7bb9aa3f Update AppImage beta build (2026-07-16 20:56:36) 2026-07-16 20:56:36 +00:00
MacRimi bcb8ec0f81 update roxmenux-monitor-v2 2026-07-16 22:53:57 +02:00
ProxMenuxBot e0fd85faac Update helpers_cache.json 2026-07-16 18:58:50 +00:00
ProxMenuxBot 0b01bad764 Update helpers_cache.json 2026-07-16 13:25:15 +00:00
MacRimiandGitHub 032ec197d6 Merge pull request #254 from MacRimi/hotfix/scheduler-pbs-attached-encryption
hotfix: prompt PBS encryption in attached-mode scheduled jobs + reorder backend menu
2026-07-15 23:45:32 +02:00
MacRimi 2fdf94f322 hotfix: prompt PBS encryption in attached-mode scheduled jobs + reorder backend menu
Two changes, both scoped to scripts/backup_restore/backup_scheduler.sh, worth
shipping to main ahead of the full v1.2.3 release PR:

1. Attached-mode PBS jobs never asked about encryption. `_create_job_attached`
   ran `hb_select_pbs_repository` and jumped straight to writing the .env with
   PBS_REPOSITORY / PBS_PASSWORD / PBS_BACKUP_ID — no `hb_ask_pbs_encryption`
   call, no PBS_KEYFILE / PBS_ENCRYPTION_PASSWORD emitted. The runner then
   invoked `proxmox-backup-client backup` without `--keyfile`, so every
   attached-mode backup landed on PBS unencrypted regardless of what the
   operator would have picked. Confirmed via `git show` on eight historical
   commits back to 61ff665c (beta 1.2.2.2) — the encryption call has NEVER
   been in the attached branch; the standalone `_create_job_new` branch had
   it since day one, they just diverged silently.

   Fix mirrors `_create_job_new` exactly (lines 413-443):
     hb_ask_pbs_encryption || return 1               # abort on cancel
     local pbs_kf_val=""
     [[ -n "${HB_PBS_KEYFILE_OPT:-}" ]] && pbs_kf_val="$HB_STATE_DIR/pbs-key.conf"
     lines+=(... "PBS_KEYFILE=${pbs_kf_val}" "PBS_ENCRYPTION_PASSWORD=${HB_PBS_ENC_PASS:-}")

   The Monitor Web path (flask_server.py::api_host_backups_job_create)
   already accepted pbs_encrypt_mode for both modes and passed it through
   correctly, so the fix is confined to the CLI wizard.

2. Backend selection menu reordered from `local | borg | pbs` to
   `pbs (recommended) | borg | local` so the recommended default sits at the
   top of the list. PBS gets the "(recommended)" suffix in its label.

Deployed and verified on the four test hosts (.50, .55, .89, .1.10) —
attached-mode wizard now shows the encryption dialog immediately after the
PBS job picker, and the .env carries PBS_KEYFILE + PBS_ENCRYPTION_PASSWORD
when the operator opts in.
2026-07-15 23:44:16 +02:00
MacRimi ed9b027c19 Update backup_scheduler.sh 2026-07-15 23:39:54 +02:00
github-actions[bot] dee9d4ae40 Update AppImage release build (2026-07-15 15:41:08) 2026-07-15 15:41:08 +00:00
MacRimiandGitHub 4ac112ff39 Merge pull request #252 from MacRimi/develop
Release 1.2.3
2026-07-15 17:36:17 +02:00
MacRimi 3915b219cc Merge branch 'main' into develop — resolve PBS page.tsx conflict (keep em helper fix) 2026-07-15 17:31:20 +02:00
MacRimi ddd9e35c83 change-language.json 2026-07-15 17:28:10 +02:00
MacRimi 892a90fa3c new version 1.2.3 2026-07-15 17:12:29 +02:00
MacRimi bbebef6929 update web 2026-07-15 17:09:02 +02:00
MacRimi 1b992988eb Update changelog 2026-07-15 16:22:58 +02:00
MacRimi 4fe335224d Delete ProxMenux-1.2.2.3-beta.AppImage 2026-07-14 19:51:13 +02:00
github-actions[bot] b4dddebd6c Update AppImage beta build (2026-07-14 17:47:14) 2026-07-14 17:47:14 +00:00
MacRimi 8060f59d69 Update version 1.2.3 2026-07-14 19:22:33 +02:00
MacRimi 1872a309ec update version 1.2.3 2026-07-13 23:19:24 +02:00
MacRimi 1b623cd275 updage installer 2026-07-13 22:03:25 +02:00
MacRimi 362409e654 update menus 2026-07-13 21:43:48 +02:00
MacRimi e771a11441 Update es.json 2026-07-13 21:20:18 +02:00
MacRimi c43bf19bae Update es.json 2026-07-13 21:16:59 +02:00
MacRimi a1167dad4b Update hw_grafics_menu.sh 2026-07-13 21:10:17 +02:00
MacRimi 63a6378f2d Update menus 2026-07-13 20:47:14 +02:00
MacRimi fffeb2c06c Update es.json 2026-07-13 20:32:40 +02:00
ProxMenuxBot ebc668077f chore(lang): auto-rebuild translation cache
Source: f5c72f1
Triggered by: push
2026-07-12 10:53:23 +00:00
MacRimi f5c72f19b8 update 1.2.2.3 beta 2026-07-12 12:52:16 +02:00
MacRimi aa3714c2ed Update pbs-paired-backup-groups.png 2026-07-08 19:59:13 +02:00
MacRimi 837d5cd95a Update pbs.json 2026-07-08 19:58:07 +02:00
MacRimi 17246ebdde update documentation 2026-07-08 19:42:48 +02:00
MacRimi 87454b5b2d Update 1.2.2.3 beta 2026-07-08 19:17:13 +02:00
MacRimi 20f85001d5 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-07-08 18:02:36 +02:00
MacRimi 71780fccf4 Update backup_host.sh 2026-07-08 18:02:26 +02:00
ProxMenuxBot c0e7a96406 chore(lang): auto-rebuild translation cache
Source: 1f5fc9a
Triggered by: push
2026-07-08 15:51:17 +00:00
MacRimi 1f5fc9ad88 Update backup_host.sh 2026-07-08 17:50:14 +02:00
ProxMenuxBot 65a5755120 chore(lang): auto-rebuild translation cache
Source: 5e368cd
Triggered by: push
2026-07-08 13:47:29 +00:00
MacRimi 5e368cd8b3 Update vzdump-hook.sh 2026-07-08 15:46:12 +02:00
MacRimi 77ae299512 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-07-08 15:37:08 +02:00
MacRimi 1b6776a53f Update lib_host_backup_common.sh 2026-07-08 15:36:55 +02:00
ProxMenuxBot d57924e25a chore(lang): auto-rebuild translation cache
Source: cc71d14
Triggered by: push
2026-07-08 13:32:19 +00:00
MacRimi cc71d14dfa Update lib_host_backup_common.sh 2026-07-08 15:30:53 +02:00
ProxMenuxBot 197998d231 chore(lang): auto-rebuild translation cache
Source: 8b6fdcf
Triggered by: push
2026-07-08 13:25:07 +00:00
MacRimi 8b6fdcf9e1 update 1.2.2.3 beta 2026-07-08 15:20:38 +02:00
MacRimi 04a1cffe7f Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-07-08 12:46:46 +02:00
MacRimi a2a0fb4705 update docs 2026-07-08 12:46:36 +02:00
ProxMenuxBot 13435a8b41 chore(lang): auto-rebuild translation cache
Source: 341681a
Triggered by: push
2026-07-08 10:22:21 +00:00
MacRimi 341681a3dc Update 1.2.2.3 beta 2026-07-08 12:18:52 +02:00
MacRimi 4789371f4d update documentation 2026-07-06 18:48:25 +02:00
MacRimi 6f0fc68c3d Update 1.2.2.3 beta 2026-07-06 17:15:22 +02:00
MacRimi 63f82d971d update 1.2.2.3 beta 2026-07-06 12:00:43 +02:00
MacRimiandGitHub a037cd15af Update community scripts to use sourced utils 2026-07-06 08:41:20 +02:00
ProxMenuxBot 64c8f4fbb3 chore(lang): auto-rebuild translation cache
Source: 9c90722
Triggered by: push
2026-07-05 23:01:22 +00:00
MacRimi 9c90722ef9 update 1.2.2.3 beta 2026-07-06 00:59:50 +02:00
ProxMenuxBot 9f94d72b69 chore(lang): auto-rebuild translation cache
Source: 7156af1
Triggered by: push
2026-07-05 22:40:47 +00:00
MacRimi 7156af1965 update 1.2.2.3 beta 2026-07-06 00:39:43 +02:00
ProxMenuxBot 20d10fc268 chore(lang): auto-rebuild translation cache
Source: 8df8a77
Triggered by: push
2026-07-05 22:34:54 +00:00
MacRimi 8df8a77bf1 update 1.2.2.3 beta 2026-07-06 00:30:47 +02:00
ProxMenuxBot e07c83f4a9 chore(lang): auto-rebuild translation cache
Source: 487ab04
Triggered by: push
2026-07-05 22:00:21 +00:00
MacRimi 487ab04a14 update 1.2.2.3 beta 2026-07-05 23:58:58 +02:00
MacRimi fd1aeb1ead Update run_scheduled_backup.sh 2026-07-05 23:38:12 +02:00
MacRimi e35ef38fea update 1.2.2.3 beta 2026-07-05 23:22:39 +02:00
MacRimiandClaude Opus 4.7 26b47e63d9 host-backup(pbs): fix scheduled-job encryption + surface real key import error + pass --keyfile on downloads
Three bugs against the PBS encryption flow:

1. Create-scheduled-job with encryption failed with "Recovery setup
   failed: no PBS keyfile present" whenever the operator picked
   "Generate a new keyfile" but had no keyfile installed yet. The
   frontend called /pbs-recovery/setup before creating the job, but
   the keyfile was only materialised later during job creation. The
   endpoint now generates the keyfile atomically if missing before
   building the escrow blob — same prompt-first order the CLI wizard
   applies. Existing keyfiles are still trusted and never rotated.

2. Importing a valid PBS keyfile via the Web dialog returned a
   generic "did not recognise this file as a valid PBS keyfile" that
   hid the real reason (kdf mismatch, missing passphrase, corrupt
   JSON, ...). The endpoint now attaches the stderr of
   `proxmox-backup-client key info` as `tool_output` and the frontend
   renders it verbatim inside the red banner. Also strips a leading
   UTF-8 BOM before validating so an editor-inserted BOM stops being
   silently classified as "invalid keyfile".

3. Downloading an encrypted PBS snapshot failed with "missing key —
   manifest was created with key XX:XX:..." even when the correct
   keyfile was installed at /usr/local/share/proxmenux/pbs-key.conf,
   because the restore worker invoked `proxmox-backup-client restore`
   without `--keyfile`. The flag is now passed whenever a local
   keyfile exists (PBS ignores it for unencrypted archives). On a
   fingerprint mismatch the error now appends the installed key's
   fingerprint so it can be compared side-by-side with the manifest's
   expected value — same fingerprint also exposed via
   /pbs-recovery/status for the UI.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-07-05 23:14:52 +02:00
MacRimiandClaude Opus 4.7 4b022d05f2 docs(restoring): swap live-progress figures and correct their captions
The two Details modal screenshots were displayed in the wrong order and
both captioned as post-completion snapshots. The one named `-details.png`
was actually captured mid-run (Restore in progress badge, ~2m left) and
the one named `-card.png` after completion (Restore complete badge,
0m53s duration). Reorders the figures to running-first then completed,
and rewrites the four alt/caption entries in EN and ES to match what
each image actually shows.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-07-05 18:05:25 +02:00
MacRimiandGitHub 118d271ea4 Update beta version from 1.2.2.2 to 1.2.2.3 2026-07-05 17:54:05 +02:00
MacRimiandClaude Opus 4.7 f92af374c7 docs(restoring): add second Details modal screenshot for live progress section
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-07-05 17:34:48 +02:00
MacRimi 5b22039600 Update 1.2.2.3 beta 2026-07-05 17:23:00 +02:00
ProxMenuxBot 6560d0ced7 chore(lang): auto-rebuild translation cache
Source: 8bcfcd6
Triggered by: push
2026-07-05 14:59:37 +00:00
MacRimi 8bcfcd6059 update 1.2.2.3 beta 2026-07-05 16:58:27 +02:00
ProxMenuxBot 6c1c317921 chore(lang): auto-rebuild translation cache
Source: 7fc7125
Triggered by: push
2026-07-05 14:53:32 +00:00
MacRimi 7fc7125c71 create 1.2.2.3 beta 2026-07-05 16:50:06 +02:00
ProxMenuxBot 9f03164258 chore(lang): auto-rebuild translation cache
Source: 29bca61
Triggered by: push
2026-07-05 07:32:39 +00:00
MacRimi 29bca610a0 update 1.2.2.2 beta 2026-07-05 09:30:13 +02:00
MacRimiandGitHub 790c8d2fd4 Update beta_version.txt 2026-07-04 23:38:21 +02:00
MacRimi 877f7a3d81 update 1.2.2.2 beta 2026-07-04 22:58:07 +02:00
MacRimi d59b1af8a9 update 1.2.2.2 beta 2026-07-04 22:30:10 +02:00
MacRimi 0d6c7290e5 Update 1.2.2.2 beta 2026-07-04 22:03:45 +02:00
MacRimi f768d9eff8 Update 1.2.2.2 beta 2026-07-04 21:50:29 +02:00
MacRimi 66dd3ec014 update 1.2.2.2 beta 2026-07-04 21:37:39 +02:00
ProxMenuxBot 29ebfdc324 chore(lang): auto-rebuild translation cache
Source: 17de0a5
Triggered by: push
2026-07-03 19:28:26 +00:00
MacRimi 17de0a5a77 Update 1.2.2.2 beta 2026-07-03 21:27:33 +02:00
ProxMenuxBot efc07056aa chore(lang): auto-rebuild translation cache
Source: 9a81c63
Triggered by: push
2026-07-03 17:29:14 +00:00
MacRimi 9a81c631fa Update backup_host.sh 2026-07-03 19:28:28 +02:00
ProxMenuxBot 2f01950d45 chore(lang): auto-rebuild translation cache
Source: c25441c
Triggered by: push
2026-07-03 17:19:03 +00:00
MacRimi c25441cca2 Update backup_host.sh 2026-07-03 19:17:03 +02:00
MacRimi 35cb10ff44 update 1.2.2.2 beta 2026-07-03 18:59:45 +02:00
ProxMenuxBot eb67fb9bd9 chore(lang): auto-rebuild translation cache
Source: c455d66
Triggered by: push
2026-07-02 21:54:25 +00:00
MacRimi c455d66b91 update 1.2.2.2 beta 2026-07-02 23:52:04 +02:00
MacRimi 153aae659c Update lib_host_backup_common.sh 2026-07-02 22:23:21 +02:00
MacRimi a00a6e5f2b Update backup_host.sh 2026-07-02 21:50:48 +02:00
ProxMenuxBot 2eef425f41 chore(lang): auto-rebuild translation cache
Source: ce3ad61
Triggered by: push
2026-07-02 19:15:15 +00:00
MacRimi ce3ad61f8f Update lib_host_backup_common.sh 2026-07-02 21:14:17 +02:00
MacRimi 0ad84572a7 Update lib_host_backup_common.sh 2026-07-02 21:11:54 +02:00
MacRimi c6fb7182e9 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-07-02 21:04:10 +02:00
MacRimi 003ae3d31a Update lib_host_backup_common.sh 2026-07-02 21:03:54 +02:00
ProxMenuxBot f17a83a1b3 chore(lang): auto-rebuild translation cache
Source: 827c88d
Triggered by: push
2026-07-02 18:58:09 +00:00
MacRimi 827c88d24f update 1.2.2.2 beta 2026-07-02 20:57:11 +02:00
github-actions[bot] 872f79a9ea Update AppImage beta build (2026-07-02 18:18:20) 2026-07-02 18:18:20 +00:00
ProxMenuxBot 0b2879ead0 chore(lang): auto-rebuild translation cache
Source: 357b2e8
Triggered by: push
2026-07-02 18:12:19 +00:00
MacRimi 357b2e8ac0 update 1.2.2.2 beta 2026-07-02 20:07:05 +02:00
github-actions[bot] efa84b0fa1 Update AppImage beta build (2026-07-02 16:24:06) 2026-07-02 16:24:06 +00:00
ProxMenuxBot 8705f638d5 chore(lang): auto-rebuild translation cache
Source: bc3c771
Triggered by: push
2026-07-02 16:15:13 +00:00
MacRimi bc3c771137 update 1.2.2.2 beta 2026-07-02 18:12:13 +02:00
github-actions[bot] f0e79e93b6 Update AppImage beta build (2026-07-01 18:57:10) 2026-07-01 18:57:10 +00:00
MacRimi f2b0b1b039 Update notification_templates.py 2026-07-01 20:54:52 +02:00
github-actions[bot] 8235549e4f Update AppImage beta build (2026-07-01 18:19:00) 2026-07-01 18:19:00 +00:00
MacRimi 6b173c42b6 Update 1.2.2.2 beta 2026-07-01 20:13:59 +02:00
github-actions[bot] d522b9a337 Update AppImage beta build (2026-06-30 16:05:14) 2026-06-30 16:05:14 +00:00
MacRimi 33a8f4baa5 update 1.2.2.2 beta 2026-06-30 17:58:32 +02:00
MacRimi b2753be204 update 1.2.2.2 beta 2026-06-28 15:17:30 +02:00
ProxMenuxBot a3d282f33e chore(lang): auto-rebuild translation cache
Source: 51b9285
Triggered by: push
2026-06-28 11:08:04 +00:00
MacRimi 51b9285980 update 1.2.2.2 beta 2026-06-28 13:06:40 +02:00
MacRimi ecfdcf1bac update 1.2.2.2 beta 2026-06-26 11:11:39 +02:00
MacRimi 63b9d69f3f update 1.2.2.2 beta 2026-06-26 10:25:06 +02:00
MacRimi a56afecccf update 1.2.2.2 beta 2026-06-25 17:29:55 +02:00
MacRimi 484f0ce897 update 1.2.2.2 beta 2026-06-25 17:07:58 +02:00
MacRimi cdb4522e5f update 1.2.2.2 beta 2026-06-25 00:26:03 +02:00
MacRimi 4f2494e135 Update 1.2.2.2 beta 2026-06-25 00:00:12 +02:00
MacRimi 202068124b update 1.2.2.2 beta 2026-06-24 23:40:22 +02:00
MacRimi 87a29f324b Update 1.2.2.2 beta 2026-06-24 21:58:21 +02:00
MacRimi 61b9fd12bb update 1.2.2.2 beta 2026-06-24 18:23:16 +02:00
MacRimi b7380fd582 Update lib_host_backup_common.sh 2026-06-24 16:14:33 +02:00
MacRimi cd2a075fab update 1.2.2.2 beta 2026-06-24 16:02:35 +02:00
MacRimi 93553574b3 Update 1.2.2.2 beta 2026-06-23 11:19:04 +02:00
MacRimi 6ab9d4ca27 Update 1.2.2.2 beta 2026-06-22 18:52:26 +02:00
MacRimi cfdd78244d Update 1.2.2.2 beta 2026-06-22 17:52:20 +02:00
MacRimi 194523c13a Update backup_host.sh 2026-06-22 17:41:34 +02:00
MacRimi 84b53fe64c Update es.json 2026-06-22 17:29:08 +02:00
MacRimi 99e0227fec Update beta_version.txt 2026-06-22 17:26:02 +02:00
ProxMenuxBot 00ecf0497c chore(lang): auto-rebuild translation cache
Source: beb3e7e
Triggered by: push
2026-06-22 15:09:41 +00:00
MacRimi beb3e7e0c4 Update 1.2.2.2 beta 2026-06-22 17:08:31 +02:00
MacRimi 3b365f8ad5 Update proxmenux_debug.sh 2026-06-22 12:55:42 +02:00
MacRimi b8038f4d41 Update proxmenux_debug.sh 2026-06-22 12:25:49 +02:00
MacRimi 68c8c03642 update 1.2.2.2 pre-beta 2026-06-22 11:38:06 +02:00
MacRimi c4cab77319 update 1.2.2.2 beta 2026-06-22 01:16:49 +02:00
MacRimi 9d099ba358 Update 1.2.2.2 beta 2026-06-22 00:55:34 +02:00
ProxMenuxBot 62fd69b147 chore(lang): auto-rebuild translation cache
Source: f50265a
Triggered by: push
2026-06-21 22:45:01 +00:00
MacRimi f50265a212 Update backup_host.sh 2026-06-22 00:44:02 +02:00
MacRimi e5669dd982 update 1.2.2.2 beta 2026-06-22 00:26:31 +02:00
MacRimi 7e6022da59 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-06-22 00:14:57 +02:00
MacRimi d807c9f085 Update flask_server.py 2026-06-22 00:14:46 +02:00
ProxMenuxBot d9a56ae0f1 chore(lang): auto-rebuild translation cache
Source: a9661a7
Triggered by: push
2026-06-21 21:52:11 +00:00
MacRimi a9661a71ff update 1.2.2.2 beta 2026-06-21 23:49:41 +02:00
MacRimi 7080570b43 update 1.2.2.2 beta 2026-06-21 23:20:00 +02:00
MacRimi a47198b37f Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-06-21 23:08:30 +02:00
MacRimi cab9c81c63 Update nvidia_installer.sh 2026-06-21 23:08:17 +02:00
ProxMenuxBot 25612b2252 chore(lang): auto-rebuild translation cache
Source: 09a67bf
Triggered by: push
2026-06-21 20:44:40 +00:00
MacRimi 09a67bf4e6 update 1.2.2.2 beta 2026-06-21 22:41:54 +02:00
MacRimi 8e92df5bd7 update 1.2.2.2 beta 2026-06-21 00:35:22 +02:00
MacRimi 57e936785d update 1.2.2.2 beta 2026-06-20 21:59:19 +02:00
MacRimi 3fd7f4b2a4 Update 1.2.2.2 beta 2026-06-19 23:38:57 +02:00
MacRimi 22a8cbc402 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-06-13 19:05:11 +02:00
MacRimi c4447eae5e update 1.2.2.2 beta 2026-06-13 19:05:00 +02:00
ProxMenuxBot d88d0f765e chore(lang): auto-rebuild translation cache
Source: 7ea9f10
Triggered by: push
2026-06-13 16:54:49 +00:00
MacRimi 7ea9f10d6f Update 1.2.2.2 beta 2026-06-13 18:52:28 +02:00
MacRimi 024ca83afd Update es.json 2026-06-13 18:29:03 +02:00
ProxMenuxBot 2f01c815e8 chore(lang): auto-rebuild translation cache
Source: 4c65d5a
Triggered by: push
2026-06-13 15:38:26 +00:00
MacRimi 4c65d5a07a update 1.2.2.2 beta 2026-06-13 17:34:11 +02:00
MacRimi 905fa4afce Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-06-13 17:11:49 +02:00
MacRimi 217d9735a6 Update es.json 2026-06-13 17:09:09 +02:00
ProxMenuxBot e3e9d740ff chore(lang): auto-rebuild translation cache
Source: 0920659
Triggered by: push
2026-06-13 14:44:52 +00:00
MacRimi 0920659693 Update commands_share.sh 2026-06-13 16:44:10 +02:00
MacRimi ae68d59a20 Update es.json 2026-06-13 16:39:39 +02:00
MacRimi 8b5d7c65d9 Update es.json 2026-06-13 16:30:05 +02:00
MacRimi b45f032f89 Update es.json 2026-06-13 16:19:59 +02:00
MacRimi ee58c21fef Update select_windows_iso.sh 2026-06-13 11:35:48 +02:00
MacRimi 2a198db593 Update select_linux_iso.sh 2026-06-13 11:34:12 +02:00
MacRimi f8be7b06d7 Update iso_storage_helpers.sh 2026-06-13 11:31:54 +02:00
MacRimi 70ed3c6f1f Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-06-13 11:20:45 +02:00
MacRimi 64356965fb Update 1.2.2.2 beta 2026-06-13 11:20:33 +02:00
ProxMenuxBot 75dcb8935d chore(lang): auto-rebuild translation cache
Source: d66dc07
Triggered by: push
2026-06-13 09:14:41 +00:00
MacRimi d66dc07ae1 Update 1.2.2.2 bate 2026-06-13 11:03:48 +02:00
MacRimi d9fee64c35 Update backup_host.sh 2026-06-12 23:58:21 +02:00
ProxMenuxBot 9039064652 chore(lang): auto-rebuild translation cache
Source: e3ce042
Triggered by: push
2026-06-12 21:54:17 +00:00
MacRimi e3ce042be4 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-06-12 23:53:13 +02:00
MacRimi c37d3fa34e update 1.2.2.2 beta 2026-06-12 23:53:08 +02:00
ProxMenuxBot a592eb896f chore(lang): auto-rebuild translation cache
Source: 2d3c8f5
Triggered by: push
2026-06-12 21:18:46 +00:00
MacRimi 2d3c8f5713 Update 1.2.2.2 beta 2026-06-12 23:17:11 +02:00
MacRimi c6d93278cd update 1.2.2.2 beta 2026-06-12 22:06:26 +02:00
MacRimi 03b6f25e14 Update nvidia_installer.sh 2026-06-12 21:03:01 +02:00
MacRimi 6ccb54e64a Update pci_passthrough_helpers.sh 2026-06-12 20:57:45 +02:00
MacRimi 6e1e47d9fd update 1.2.2.2 beta 2026-06-12 20:05:17 +02:00
MacRimi 761357b737 Update 1.2.2.2 beta 2026-06-12 19:57:54 +02:00
MacRimi 57f1ebc358 Update lib_host_backup_common.sh 2026-06-12 18:28:06 +02:00
MacRimi 345d66e0fd Update nvidia_installer.sh 2026-06-12 17:19:20 +02:00
MacRimi b5cdf1d2a6 Update apply_cluster_postboot.sh 2026-06-12 17:04:56 +02:00
MacRimi 382493ca84 update 1.2.2.2 beta 2026-06-12 00:05:58 +02:00
ProxMenuxBot 450fceec62 chore(lang): auto-rebuild translation cache
Source: 41de70a
Triggered by: push
2026-06-11 21:49:48 +00:00
MacRimi 41de70a1b9 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-06-11 23:49:16 +02:00
MacRimi 9bedb4f311 update 1.2.2.2 beta 2026-06-11 23:49:13 +02:00
ProxMenuxBot 7166a10ac2 chore(lang): auto-rebuild translation cache
Source: ed924b6
Triggered by: push
2026-06-11 21:10:52 +00:00
MacRimi ed924b67fe update 1.2.2.2 beta 2026-06-11 23:08:56 +02:00
ProxMenuxBot a4d1e9fbb9 chore(lang): auto-rebuild translation cache
Source: 9afbf0e
Triggered by: push
2026-06-11 17:16:13 +00:00
MacRimi 9afbf0ea5e Update lib_host_backup_common.sh 2026-06-11 19:11:23 +02:00
ProxMenuxBot 3d9ade0f37 chore(lang): auto-rebuild translation cache
Source: 6094ab8
Triggered by: push
2026-06-11 15:26:13 +00:00
MacRimi 6094ab8e1c update beta 1.2.2.2 2026-06-11 17:24:20 +02:00
MacRimi f9cf931828 Update run_scheduled_backup.sh 2026-06-10 20:10:26 +02:00
MacRimi 7a88971114 Merge branch 'develop' of https://github.com/MacRimi/ProxMenux into develop 2026-06-10 20:00:08 +02:00
MacRimi 827aa7154f update 1.2.2.2 beta 2026-06-10 19:59:56 +02:00
ProxMenuxBot 1ebf4d0a28 chore(lang): auto-rebuild translation cache
Source: df95b50
Triggered by: push
2026-06-10 17:55:16 +00:00
MacRimi df95b50f8c Update beta 1.2.2.2 2026-06-10 19:53:40 +02:00
MacRimi 7ad5508623 lang: seed translation cache (es, fr, de, it, pt) 2026-06-10 19:12:45 +02:00
MacRimi 4dc8be7387 Add beta 1.2.2.2 2026-06-10 19:05:13 +02:00
MacRimi 165e8c9636 Update backup_host.sh 2026-06-09 19:44:14 +02:00
MacRimi cff2ca3c95 Update lib_host_backup_common.sh 2026-06-09 19:17:20 +02:00
MacRimi d41871cc53 update 1.2.2.1 beta 2026-06-09 19:14:27 +02:00
MacRimi 6b3c42e0ed Delete test_backup_restore.sh 2026-06-09 17:58:52 +02:00
MacRimi d41eaef8a2 delete files backups scripts 2026-06-09 17:56:17 +02:00
MacRimi f54118843e Create jc_channel.txt 2026-06-09 17:48:23 +02:00
MacRimi f0b8474350 Update 1.2.2.1 beta 2026-06-09 17:42:51 +02:00
MacRimi 61ff665cec update beta 1.2.2.2 2026-06-09 00:13:24 +02:00
MacRimi 6844406cf7 Update 1.2.2.1 2026-06-07 11:31:50 +02:00
MacRimi 61ff98e830 Update beta 1.2.2.1 2026-06-06 18:30:11 +02:00
MacRimi 66419777d8 Update beta 1.2.2.1 2026-06-06 11:37:54 +02:00
MacRimi d401e5f7de Add new beta 1.2.2.1 2026-06-05 19:45:46 +02:00
MacRimi 3191f5250d Update 1.2.2.1 beta 2026-06-05 19:22:07 +02:00
202 changed files with 68687 additions and 9608 deletions
File diff suppressed because it is too large Load Diff
+388
View File
@@ -0,0 +1,388 @@
#!/usr/bin/env python3
"""
Build the ProxMenux translation cache from translate calls in scripts/.
The generated JSON keeps the same shape used by scripts/utils.sh:
{
"Original English text": {
"es": "Translated text",
"fr": "Translated text"
}
}
"""
from __future__ import annotations
import argparse
import ast
import json
import os
import subprocess
import re
import sys
import time
from pathlib import Path
from typing import Iterable
from urllib.parse import quote
from urllib.request import Request, urlopen
DEFAULT_LANGUAGES = ("es", "fr", "de", "it", "pt")
DEFAULT_CONTEXT = "Context: Technical message for Proxmox and IT. Translate:"
TRANSLATE_CALL_RE = re.compile(
r"""translate\s+(?P<quote>["'])(?P<text>(?:\\.|(?! (?P=quote) ).)*?)(?P=quote)""",
re.VERBOSE | re.DOTALL,
)
def iter_script_files(
scripts_dir: Path, extra_files: Iterable[Path] = ()
) -> Iterable[Path]:
# Walk the main scripts tree.
for path in sorted(scripts_dir.rglob("*")):
if not path.is_file():
continue
if path.name == "utils.sh":
continue
if path.suffix not in {".sh", ".func"}:
continue
yield path
# Yield additional files passed explicitly (e.g. the root-level `menu`
# entry point or install_proxmenux*.sh). These live outside scripts/
# but still contain translate "..." calls we want in the cache.
# No extension filter and no utils.sh skip — the caller decided
# they belong, we just check the file actually exists.
for extra in extra_files:
if extra.is_file():
yield extra
def decode_shell_string(raw: str, quote_char: str) -> str:
if quote_char == "'":
return raw
try:
return ast.literal_eval(f'"{raw}"')
except Exception:
return raw.replace(r"\"", '"').replace(r"\\", "\\")
def extract_translate_texts(
scripts_dir: Path, extra_files: Iterable[Path] = ()
) -> list[str]:
found: dict[str, None] = {}
for path in iter_script_files(scripts_dir, extra_files):
try:
content = path.read_text(encoding="utf-8")
except UnicodeDecodeError:
content = path.read_text(encoding="utf-8", errors="replace")
for match in TRANSLATE_CALL_RE.finditer(content):
text = decode_shell_string(match.group("text"), match.group("quote"))
text = text.strip()
if text and "$" not in text and "`" not in text:
found.setdefault(text, None)
return sorted(found)
def translate_googletrans(text: str, dest_lang: str, context: str) -> str:
try:
from googletrans import Translator # type: ignore
except Exception as exc:
raise RuntimeError(
"googletrans is not installed. Install googletrans==4.0.0-rc1 "
"or run with --provider google-web."
) from exc
translator = Translator()
full_text = f"{context} {text}".strip()
return translator.translate(full_text, dest=dest_lang).text
def translate_google_web(text: str, dest_lang: str, context: str, timeout: int) -> str:
# The public Google endpoint is not prompt-aware: if we prepend context,
# it often translates and returns that context as part of the result.
full_text = text
url = (
"https://translate.googleapis.com/translate_a/single"
f"?client=gtx&sl=en&tl={quote(dest_lang)}&dt=t&q={quote(full_text)}"
)
req = Request(url, headers={"User-Agent": "ProxMenux translation cache builder"})
with urlopen(req, timeout=timeout) as response:
payload = json.loads(response.read().decode("utf-8"))
return "".join(part[0] for part in payload[0] if part and part[0])
def translate_appimage(
text: str,
dest_lang: str,
context: str,
timeout: int,
appimage_path: Path,
) -> str:
if not appimage_path.exists():
prev_path = appimage_path.with_name(appimage_path.name + ".prev")
if prev_path.exists():
appimage_path = prev_path
else:
raise FileNotFoundError(f"AppImage not found: {appimage_path}")
req = {
"text": text,
"dest_lang": dest_lang,
"context": context,
"cache_file": "",
}
env = os.environ.copy()
env.setdefault("APPIMAGE_EXTRACT_AND_RUN", "1")
completed = subprocess.run(
[str(appimage_path), "--translate"],
input=json.dumps(req, ensure_ascii=False),
text=True,
capture_output=True,
timeout=timeout,
check=False,
env=env,
)
if completed.returncode != 0:
raise RuntimeError((completed.stderr or completed.stdout).strip())
# AppRun may print a startup line before translate_cli.py emits JSON.
for line in reversed(completed.stdout.splitlines()):
line = line.strip()
if not line.startswith("{"):
continue
payload = json.loads(line)
if payload.get("success"):
return str(payload.get("text", text))
raise RuntimeError(str(payload.get("error", "unknown AppImage translation error")))
raise RuntimeError(f"AppImage did not return JSON: {completed.stdout.strip()}")
def clean_translation(value: str) -> str:
separator = r"[\s\u00a0]*[:]"
translate_labels = "Translate|Traducir|Traduire|Übersetzen|Tradurre|Traduci|Traduzir"
context_labels = "Context|Contexto|Contexte|Kontext|Contesto"
value = re.sub(
rf"^.*?({translate_labels}){separator}",
"",
value,
flags=re.IGNORECASE | re.DOTALL,
)
value = re.sub(
rf"^.*?({context_labels}){separator}.*?({translate_labels}){separator}",
"",
value,
flags=re.IGNORECASE | re.DOTALL,
)
value = re.sub(
rf"^.*?({context_labels}){separator}",
"",
value,
flags=re.IGNORECASE | re.DOTALL,
)
return value.strip()
def translate_text(
text: str,
dest_lang: str,
provider: str,
context: str,
timeout: int,
appimage_path: Path,
) -> str:
if provider == "googletrans":
translated = translate_googletrans(text, dest_lang, context)
elif provider == "google-web":
translated = translate_google_web(text, dest_lang, context, timeout)
elif provider == "appimage":
translated = translate_appimage(text, dest_lang, context, timeout, appimage_path)
else:
raise ValueError(f"Unknown provider: {provider}")
return clean_translation(translated) or text
def load_language_cache(path: Path) -> dict[str, str]:
if not path.exists():
return {}
try:
data = json.loads(path.read_text(encoding="utf-8"))
except Exception:
return {}
if not isinstance(data, dict):
return {}
return {str(text): str(value) for text, value in data.items()}
def write_language_cache(path: Path, cache: dict[str, str]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
tmp_path = path.with_suffix(path.suffix + ".tmp")
tmp_path.write_text(
json.dumps(cache, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
encoding="utf-8",
)
tmp_path.replace(path)
def build_arg_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description="Extract translate calls from scripts/ and build json/cache.json."
)
parser.add_argument("--scripts-dir", default="scripts", type=Path)
parser.add_argument(
"--extra-file",
action="append",
default=[],
type=Path,
metavar="PATH",
help=(
"Extra individual files to scan for translate calls in addition "
"to --scripts-dir. Useful for the root-level `menu` entry point "
"and install_proxmenux*.sh, which sit outside scripts/. "
"Pass multiple times to add more than one file."
),
)
parser.add_argument(
"--output-dir",
default=Path("lang"),
type=Path,
help="Directory where per-language JSON files are written. Default: lang",
)
parser.add_argument(
"--output",
default=None,
type=Path,
help="Deprecated combined cache path. If used, per-language files are written next to it under its parent directory.",
)
parser.add_argument(
"--languages",
default=",".join(DEFAULT_LANGUAGES),
help="Comma-separated destination languages. Default: es,fr,de,it,pt",
)
parser.add_argument(
"--provider",
choices=("appimage", "googletrans", "google-web"),
default="appimage",
help="Translation provider to use. Default: appimage",
)
parser.add_argument(
"--appimage-path",
default=Path("/usr/local/share/proxmenux/ProxMenux-Monitor.AppImage"),
type=Path,
help="Path to the ProxMenux AppImage when using --provider appimage.",
)
parser.add_argument("--context", default=DEFAULT_CONTEXT)
parser.add_argument("--timeout", default=30, type=int)
parser.add_argument("--sleep", default=0.15, type=float)
parser.add_argument(
"--refresh",
action="store_true",
help="Translate all entries again instead of reusing existing cache values.",
)
parser.add_argument(
"--extract-only",
action="store_true",
help="Only update the cache keys; missing translations are left empty.",
)
parser.add_argument(
"--limit",
type=int,
default=0,
help="Only process the first N extracted strings. Useful for test runs.",
)
parser.add_argument(
"--save-every",
type=int,
default=1,
help="Write the output JSON every N translated items. Default: 1",
)
return parser
def main() -> int:
args = build_arg_parser().parse_args()
scripts_dir = args.scripts_dir.resolve()
if args.output is not None:
output_dir = args.output.resolve().parent / "lang"
else:
output_dir = args.output_dir.resolve()
languages = [lang.strip() for lang in args.languages.split(",") if lang.strip()]
if not scripts_dir.is_dir():
print(f"Scripts directory not found: {scripts_dir}", file=sys.stderr)
return 1
if not languages:
print("No destination languages selected.", file=sys.stderr)
return 1
texts = extract_translate_texts(scripts_dir, args.extra_file)
if args.limit > 0:
texts = texts[: args.limit]
existing_by_lang = {
lang: load_language_cache(output_dir / f"{lang}.json")
for lang in languages
}
next_by_lang: dict[str, dict[str, str]] = {lang: {} for lang in languages}
print(f"Found {len(texts)} unique translate strings.", flush=True)
print(f"Output directory: {output_dir}", flush=True)
print(f"Languages: {', '.join(languages)}", flush=True)
failures: list[tuple[str, str, str]] = []
total = len(texts) * len(languages)
done = 0
for lang in languages:
existing = existing_by_lang.get(lang, {})
print(f"Starting language: {lang}", flush=True)
for index, text in enumerate(texts, start=1):
done += 1
if not args.refresh and existing.get(text):
next_by_lang[lang][text] = existing[text]
continue
if args.extract_only:
next_by_lang[lang][text] = existing.get(text, "")
continue
print(f"[{done}/{total}] {lang} ({index}/{len(texts)}): {text[:80]}", flush=True)
try:
next_by_lang[lang][text] = translate_text(
text,
lang,
args.provider,
args.context,
args.timeout,
args.appimage_path,
)
print(f" => {next_by_lang[lang][text][:100]}", flush=True)
except Exception as exc:
next_by_lang[lang][text] = existing.get(text, text)
failures.append((text, lang, str(exc)))
print(f" failed: {exc}", file=sys.stderr, flush=True)
if args.save_every > 0 and index % args.save_every == 0:
write_language_cache(output_dir / f"{lang}.json", next_by_lang[lang])
time.sleep(args.sleep)
write_language_cache(output_dir / f"{lang}.json", next_by_lang[lang])
print(f"Completed language: {lang}", flush=True)
for lang, cache in next_by_lang.items():
write_language_cache(output_dir / f"{lang}.json", cache)
if failures:
print(f"Completed with {len(failures)} translation failures.", file=sys.stderr, flush=True)
for text, lang, error in failures[:20]:
print(f"- {lang}: {text[:80]} -> {error}", file=sys.stderr, flush=True)
if len(failures) > 20:
print(f"... and {len(failures) - 20} more.", file=sys.stderr, flush=True)
return 2
print("Translation cache generated successfully.", flush=True)
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -0,0 +1,99 @@
name: Build translation cache
# Regenerates lang/*.json whenever a bash script under scripts/ changes.
# The runtime translate() in scripts/utils.sh reads these JSON files and
# falls back to the English source on miss, so keeping them up-to-date is
# what makes ProxMenux multilingual without any runtime googletrans
# dependency on the user's host.
#
# Triggers:
# - push to develop touching scripts/**/*.sh
# - manual via workflow_dispatch (force --refresh)
on:
push:
branches: [develop]
paths:
- 'scripts/**/*.sh'
- 'menu'
- 'install_proxmenux.sh'
- 'install_proxmenux_beta.sh'
- '.github/scripts/build_translation_cache.py'
- '.github/workflows/build-translation-cache.yml'
workflow_dispatch:
inputs:
refresh:
description: 'Re-translate every entry (ignore cached values)'
type: boolean
default: false
# Avoid two cache rebuilds from racing each other on the same branch and
# fighting over the auto-commit.
concurrency:
group: build-translation-cache-${{ github.ref }}
cancel-in-progress: false
jobs:
rebuild-cache:
runs-on: ubuntu-latest
permissions:
contents: write # auto-commit lang/*.json back to develop
steps:
- name: Checkout develop
uses: actions/checkout@v4
with:
ref: develop
# Need full history so the auto-commit doesn't fail when the
# cache job runs minutes after the trigger push (GH default
# fetch-depth=1 sometimes diverges from origin/develop after a
# quick follow-up push).
fetch-depth: 0
# Use a PAT (or default GITHUB_TOKEN if branch protections allow
# it) so the push back to develop actually fires later steps
# (workflow runs from auto-commits) if you ever need them.
token: ${{ secrets.GITHUB_TOKEN }}
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.11'
- name: Install googletrans
run: |
python -m pip install --upgrade pip
# 4.0.0-rc1 is the same pin that build_translation_cache.py
# was written against. Bump in lockstep with the script.
pip install 'googletrans==4.0.0-rc1' 'httpx==0.13.3' 'httpcore==0.9.1' 'h11==0.9.0'
- name: Regenerate lang/*.json
run: |
REFRESH_FLAG=""
if [[ "${{ github.event.inputs.refresh }}" == "true" ]]; then
REFRESH_FLAG="--refresh"
fi
# Extra files outside scripts/ that also contain translate "..."
# calls. Keep this list in sync with the `paths` trigger above.
python .github/scripts/build_translation_cache.py \
--scripts-dir scripts \
--extra-file menu \
--extra-file install_proxmenux.sh \
--extra-file install_proxmenux_beta.sh \
--output-dir lang \
--provider googletrans \
$REFRESH_FLAG
- name: Commit + push if changed
run: |
if git diff --quiet -- lang/; then
echo "No translation changes — skipping commit."
exit 0
fi
git config user.name "ProxMenuxBot"
git config user.email "bot@proxmenux.local"
git add lang/
git commit -m "chore(lang): auto-rebuild translation cache
Source: ${GITHUB_SHA::7}
Triggered by: ${{ github.event_name }}"
git push origin develop
+26
View File
@@ -0,0 +1,26 @@
name: Update project growth
on:
schedule:
- cron: "17 4 * * *"
workflow_dispatch:
permissions:
contents: write
concurrency:
group: repo-growth
cancel-in-progress: false
jobs:
update:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7
- uses: MacRimi/repo-growth@v1
with:
title: ProxMenux growth
output: images/project-growth.svg
history: .github/repo-growth/history.json
metrics: stars,forks
layout: dashboard
Binary file not shown.
BIN
View File
Binary file not shown.
+1 -1
View File
@@ -1 +1 @@
e0128ac327ea74b645b37bcfab03aa744f067a355f9960a822b411c6ac75cb02 ProxMenux-1.2.2.AppImage
d3aedf08d50d332161a3c1f7f58f4f546af46ee2b7a4bd193ff6b2669b7cd5e8 ProxMenux-1.2.4.AppImage
+102 -34
View File
@@ -27,19 +27,97 @@ A modern, responsive dashboard for monitoring Proxmox VE systems built with Next
## Overview
**ProxMenux Monitor** is a comprehensive, real-time monitoring dashboard for Proxmox VE environments. Built with modern web technologies, it provides an intuitive interface to monitor system resources, virtual machines, containers, storage, network traffic, and system logs.
**ProxMenux Monitor** is a comprehensive, real-time monitoring dashboard for Proxmox VE environments. Built with modern web technologies, it provides an intuitive interface to monitor system resources, virtual machines, containers, storage, network traffic, backups, health status and system logs — all from a single browser tab.
The application runs as a standalone AppImage on your Proxmox server and serves a web interface accessible from any device on your network.
**Full documentation:** [proxmenux.com/docs/monitor](https://proxmenux.com/docs/monitor) — per-feature walkthroughs, API reference and integration guides.
## Screenshots
Get a quick overview of ProxMenux Monitor's main features:
A tour of the Monitor's main areas. See the [public documentation](https://proxmenux.com/docs/monitor) for the full walkthrough.
### Dashboard overview
<p align="center">
<img src="public/images/onboarding/imagen1.png" alt="Overview Dashboard" width="800"/>
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/dashboard-home.png" alt="Dashboard home" width="900"/>
<br/>
<em>System Overview - Monitor CPU, memory, temperature, and uptime in real-time</em>
<em>Real-time CPU, memory, temperature, storage and network activity on one glance-optimised page.</em>
</p>
### Backup & Restore
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/images/docs/backup-restore/scheduled-backup-monitor.png" alt="Scheduled backup jobs" width="900"/>
<br/>
<em>Integrated host backup & restore — Local, PBS or Borg destinations; own timer or attached to a PVE vzdump job with live-inherited retention; PBS encryption with paired recovery blobs; direction-aware cross-kernel restore.</em>
</p>
### Storage & SMART
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/storage-top-row.png" alt="Storage top row" width="900"/>
<br/>
<em>Per-disk cards with capacity palette shared across Storage and Backups. USB-NVMe enclosures (ASMedia / JMicron / Realtek) show the drive's real identity and temperature, not the bridge.</em>
</p>
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/disk-modal-smart.png" alt="Disk SMART modal" width="900"/>
<br/>
<em>Per-disk detail modal — SMART attributes, temperature history, observations and downloadable PDF report.</em>
</p>
### Network — live topology
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/network-flow-overview.png" alt="Network Flow diagram" width="900"/>
<br/>
<em>Network Flow — live view of NICs → host → bridges → guests with animated rx / tx pulses on every link. Bridges without active guests are hidden.</em>
</p>
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/network-latency-historical.png" alt="Network latency historical" width="900"/>
<br/>
<em>Latency modal — historical view and real-time ping test against Gateway / Cloudflare / Google, with a downloadable PDF report.</em>
</p>
### Virtual machines & LXCs
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/vms-top-row.png" alt="VMs & LXCs overview" width="900"/>
<br/>
<em>Inventory of running VMs and containers with resource usage and per-guest controls.</em>
</p>
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/vms-modal-status.png" alt="VM detail modal — status" width="900"/>
<br/>
<em>Per-guest modal with status, backups, mounts and (for LXC) apt / apk / community-scripts update inventory.</em>
</p>
### Health Monitor
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/health-monitor.png" alt="Health Monitor" width="900"/>
<br/>
<em>Ten categories of proactive health checks with hysteresis, per-error dismiss (24 h / 7 d / permanent) and an Active Suppressions panel to revoke.</em>
</p>
### System overview — top processes
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/system-overview-top-processes.png" alt="Top processes" width="900"/>
<br/>
<em>Per-process CPU / memory / I/O sorted by consumer, with a click-through detail modal.</em>
</p>
### Mobile
<p align="center">
<img src="https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/mobile-home.png" alt="Mobile responsive layout" width="360"/>
<br/>
<em>Fully responsive — every panel adapts to phone and tablet layouts.</em>
</p>
@@ -47,18 +125,21 @@ Get a quick overview of ProxMenux Monitor's main features:
## Features
- **System Overview**: Real-time monitoring of CPU, memory, temperature, and system uptime
- **Storage Management**: Visual representation of storage distribution, disk health, and SMART data
- **Network Monitoring**: Network interface statistics, real-time traffic graphs, and bandwidth usage
- **Virtual Machines & LXC**: Comprehensive view of all VMs and containers with resource usage and controls
- **Hardware Information**: Detailed hardware specifications including CPU, GPU, PCIe devices, and disks
- **System Logs**: Real-time system log monitoring with filtering and search capabilities
- **Health Monitoring**: Proactive system health checks with persistent error tracking
- **Authentication & 2FA**: Optional password protection with TOTP-based two-factor authentication
- **RESTful API**: Complete API access for integrations with Homepage, Home Assistant, and custom dashboards
- **Dark/Light Theme**: Toggle between themes with Proxmox-inspired design
- **Responsive Design**: Works seamlessly on desktop, tablet, and mobile devices
- **Release Notes**: Automatic notifications of new features and improvements
- **Host Backup & Restore** — integrated section covering Local, PBS and Borg destinations; schedule with a systemd timer or attach to an existing PVE vzdump job; PBS encryption with paired recovery blobs; direction-aware cross-kernel restore with kernel-agnostic hydration (IOMMU / VFIO / GRUB), cascade-safe package replay, and NIC auto-remap by MAC after a motherboard swap
- **System Overview** — real-time CPU / memory / temperature / uptime with a per-process drill-down (Top Processes view) and click-through detail modal
- **Storage & SMART** — per-disk cards in a responsive grid; temperature history, SMART attributes and a downloadable PDF SMART report per drive. USB-NVMe bridges (ASMedia / JMicron / Realtek) show the real drive's identity and temperature instead of the enclosure's
- **Network** — live topology (Network Flow) with animated rx / tx pulses, per-interface RRD charts, latency modal against Gateway / Cloudflare / Google with a downloadable PDF report
- **Virtual Machines & LXCs** — inventory, per-guest metrics and controls, per-LXC update inventory (apt / apk / community-scripts), backup state, mount inspection
- **Hardware** — CPU / GPU / PCIe / disks / NICs, with correct SSD vs HDD classification even behind USB-SATA bridges
- **System Logs** — journalctl with severity / since / free-text filters and a download-as-text action
- **Health Monitor** — ten categories of proactive checks with hysteresis and configurable thresholds; per-error dismiss (24 h / 7 d / permanent) and an Active Suppressions panel to revoke
- **Notifications** — five channels (Telegram, Discord, Gotify, Email, Apprise for ~80 endpoints), per-event toggles, Quiet Hours, Daily Digest, optional AI enrichment (Groq / OpenAI / Ollama / Gemini / Anthropic / OpenRouter)
- **Authentication & 2FA** — optional password protection with TOTP-based two-factor authentication and long-lived API tokens (365 days) for integrations
- **RESTful API** — complete access with JWT auth; ready-made recipes for Homepage, Home Assistant and custom dashboards; Prometheus scrape endpoint
- **Reports (PDF)** — SMART report per-disk and Network Latency report per-target, ready to send to a vendor or an ISP
- **Dark / Light Theme** — Proxmox-inspired palette with live switch
- **Responsive Design** — desktop, tablet and mobile layouts
- **Release Notes** — automatic notification of new features on each Monitor upgrade
## Technology Stack
@@ -106,7 +187,7 @@ On first launch, you'll be presented with three options:
2. **Enable 2FA** - Add TOTP-based two-factor authentication for enhanced security
3. **Skip** - Continue without authentication (not recommended for production environments)
![Authentication Setup](AppImage/public/images/docs/auth-setup.png)
![Authentication Setup](https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/auth-setup.png)
### Two-Factor Authentication (2FA)
@@ -118,7 +199,7 @@ After setting up your password, you can enable 2FA using any TOTP authenticator
4. Enter the 6-digit code to verify
5. Save your backup codes in a secure location
![2FA Setup](AppImage/public/images/docs/2fa-setup.png)
![2FA Setup](https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/2fa-setup.png)
### Security Best Practices for API Tokens
@@ -254,7 +335,7 @@ The easiest way to generate an API token is through the ProxMenux Monitor web in
6. Click **Generate Token**
7. Copy the token immediately - it will not be shown again
![Generate API Token](AppImage/public/images/docs/generate-api-token.png)
![Generate API Token](https://raw.githubusercontent.com/MacRimi/ProxMenux/main/web/public/monitor/api-tokens.png)
The token will be valid for **365 days** (1 year) and can be used for integrations with Homepage, Home Assistant, or any custom application.
@@ -643,8 +724,6 @@ Finally, reference the secret in your `services.yaml`:
format: bytes
```
![Homepage Integration Example](AppImage/public/images/docs/homepage-integration.png)
### Home Assistant Integration
[Home Assistant](https://www.home-assistant.io/) is an open-source home automation platform.
@@ -727,24 +806,13 @@ entities:
icon: mdi:clock-outline
```
![Home Assistant Integration Example](AppImage/public/images/docs/homeassistant-integration.png)
---
## License
This project is licensed under the **Creative Commons Attribution-NonCommercial 4.0 International License (CC BY-NC 4.0)**.
This project is licensed under the **GNU General Public License, version 3 (GPL-3.0)**.
You are free to:
- Share — copy and redistribute the material in any medium or format
- Adapt — remix, transform, and build upon the material
Under the following terms:
- Attribution — You must give appropriate credit, provide a link to the license, and indicate if changes were made
- NonCommercial — You may not use the material for commercial purposes
For more details, see the [full license](https://creativecommons.org/licenses/by-nc/4.0/).
You are free to use, study, share and modify the software under the terms of the licence. Any distributed derivative work must be licensed under the same terms and include the full source code — see the [full licence text](https://www.gnu.org/licenses/gpl-3.0.html) or the [`LICENSE`](https://github.com/MacRimi/ProxMenux/blob/main/LICENSE) file at the repository root.
+4
View File
@@ -3,6 +3,8 @@ import type { Metadata, Viewport } from "next"
import { GeistSans } from "geist/font/sans"
import { GeistMono } from "geist/font/mono"
import { ThemeProvider } from "../components/theme-provider"
import { PwaRegister } from "../components/pwa-register"
import { PwaInstallPrompt } from "../components/pwa-install-prompt"
import { Suspense } from "react"
import "./globals.css"
@@ -46,6 +48,8 @@ export default function RootLayout({
{children}
</ThemeProvider>
</Suspense>
<PwaRegister />
<PwaInstallPrompt />
</body>
</html>
)
+17 -2
View File
@@ -3,7 +3,7 @@
import { useEffect, useRef, useState } from "react"
import { Thermometer } from "lucide-react"
import { Badge } from "./ui/badge"
import { AreaChart, Area, ResponsiveContainer, Tooltip } from "recharts"
import { AreaChart, Area, ResponsiveContainer, Tooltip, YAxis } from "recharts"
import { fetchApi } from "@/lib/api-config"
import { useDiskTempThresholds } from "@/lib/health-thresholds"
@@ -64,8 +64,11 @@ export function DiskTemperatureCard({
const fetchHistory = async () => {
setLoading(true)
try {
// 24-h timeframe gives a more useful "is this drive trending
// up over a day" view; the 1-h window was too short to spot
// anything that mattered.
const result = await fetchApi<{ data: TempPoint[] }>(
`/api/disk/${encodeURIComponent(diskName)}/temperature/history?timeframe=hour`,
`/api/disk/${encodeURIComponent(diskName)}/temperature/history?timeframe=day`,
)
if (cancelled.current) return
setData(result?.data || [])
@@ -142,6 +145,18 @@ export function DiskTemperatureCard({
<stop offset="100%" stopColor={lineColor} stopOpacity={0.02} />
</linearGradient>
</defs>
{/* Y domain is computed the same way the detail modal
does — floor(min3) / ceil(max+3), floor at 0 — so
the line shape here matches the bigger chart instead
of recharts' auto-domain collapsing 24 → 29 °C into
a near-flat line that doesn't look like the modal. */}
<YAxis
hide
domain={[
(dataMin: number) => Math.max(0, Math.floor(dataMin - 3)),
(dataMax: number) => Math.ceil(dataMax + 3),
]}
/>
<Tooltip content={<MiniTooltip />} cursor={{ stroke: lineColor, strokeOpacity: 0.3, strokeWidth: 1 }} />
<Area
type="monotone"
+20 -50
View File
@@ -7,6 +7,7 @@ import { Dialog, DialogContent, DialogDescription, DialogHeader, DialogTitle } f
import { Cpu, HardDrive, Thermometer, Zap, Loader2, CpuIcon, Cpu as Gpu, Network, MemoryStick, PowerIcon, FanIcon, Battery, Usb, BrainCircuit, AlertCircle } from "lucide-react"
import { Download } from "lucide-react"
import { Button } from "@/components/ui/button"
import { getDiskType } from "../lib/disk-type"
import useSWR from "swr"
import { useState, useEffect } from "react"
import {
@@ -2460,25 +2461,10 @@ return (
)
.map((device, index) => {
const getDiskTypeBadge = (diskName: string, rotationRate: number | string | undefined) => {
let diskType = "HDD"
// Check if it's NVMe
if (diskName.startsWith("nvme")) {
diskType = "NVMe"
}
// Check rotation rate for SSD vs HDD
else if (rotationRate !== undefined && rotationRate !== null) {
// Handle both number and string formats
const rateNum = typeof rotationRate === "string" ? Number.parseInt(rotationRate) : rotationRate
if (rateNum === 0 || isNaN(rateNum)) {
diskType = "SSD"
}
}
// If rotation_rate is "Solid State Device" string
else if (typeof rotationRate === "string" && rotationRate.includes("Solid State")) {
diskType = "SSD"
}
// Classifier lives in lib/disk-type.ts — same rules
// the Storage page uses, so a drive can't show up as
// HDD here while showing as SSD there.
const diskType = getDiskType(diskName, rotationRate)
const badgeStyles: Record<string, { className: string; label: string }> = {
NVMe: {
className: "bg-purple-500/10 text-purple-500 border-purple-500/20",
@@ -2600,38 +2586,22 @@ return (
<div className="flex justify-between border-b border-border/50 pb-2">
<span className="text-sm font-medium text-muted-foreground">Type</span>
{(() => {
const getDiskTypeBadge = (diskName: string, rotationRate: number | string | undefined) => {
let diskType = "HDD"
if (diskName.startsWith("nvme")) {
diskType = "NVMe"
} else if (rotationRate !== undefined && rotationRate !== null) {
const rateNum = typeof rotationRate === "string" ? Number.parseInt(rotationRate) : rotationRate
if (rateNum === 0 || isNaN(rateNum)) {
diskType = "SSD"
}
} else if (typeof rotationRate === "string" && rotationRate.includes("Solid State")) {
diskType = "SSD"
}
const badgeStyles: Record<string, { className: string; label: string }> = {
NVMe: {
className: "bg-purple-500/10 text-purple-500 border-purple-500/20",
label: "NVMe SSD",
},
SSD: {
className: "bg-cyan-500/10 text-cyan-500 border-cyan-500/20",
label: "SSD",
},
HDD: {
className: "bg-blue-500/10 text-blue-500 border-blue-500/20",
label: "HDD",
},
}
return badgeStyles[diskType]
const diskType = getDiskType(selectedDisk.name, selectedDisk.rotation_rate)
const badgeStyles: Record<string, { className: string; label: string }> = {
NVMe: {
className: "bg-purple-500/10 text-purple-500 border-purple-500/20",
label: "NVMe SSD",
},
SSD: {
className: "bg-cyan-500/10 text-cyan-500 border-cyan-500/20",
label: "SSD",
},
HDD: {
className: "bg-blue-500/10 text-blue-500 border-blue-500/20",
label: "HDD",
},
}
const diskBadge = getDiskTypeBadge(selectedDisk.name, selectedDisk.rotation_rate)
const diskBadge = badgeStyles[diskType]
return <Badge className={diskBadge.className}>{diskBadge.label}</Badge>
})()}
</div>
+41 -4
View File
@@ -32,6 +32,7 @@ import {
FileText,
RefreshCw,
Shield,
Download,
X,
Clock,
BellOff,
@@ -39,6 +40,7 @@ import {
Settings2,
HelpCircle,
} from "lucide-react"
import { ScriptTerminalModal } from "./script-terminal-modal"
interface CategoryCheck {
status: string
@@ -122,14 +124,15 @@ export function HealthStatusModal({ open, onOpenChange, getApiUrl }: HealthStatu
const [error, setError] = useState<string | null>(null)
const [dismissingKey, setDismissingKey] = useState<string | null>(null)
const [expandedCategories, setExpandedCategories] = useState<Set<string>>(new Set())
const [showUpdateTerminal, setShowUpdateTerminal] = useState(false)
const fetchHealthDetails = useCallback(async () => {
const fetchHealthDetails = useCallback(async (force = false) => {
setLoading(true)
setError(null)
try {
let newOverallStatus = "OK"
// Use the new combined endpoint for fewer round-trips
const token = getAuthToken()
const authHeaders: Record<string, string> = {}
@@ -137,7 +140,7 @@ export function HealthStatusModal({ open, onOpenChange, getApiUrl }: HealthStatu
authHeaders["Authorization"] = `Bearer ${token}`
}
const response = await fetch(getApiUrl("/api/health/full"), { headers: authHeaders })
const response = await fetch(getApiUrl(force ? "/api/health/full?refresh=1" : "/api/health/full"), { headers: authHeaders })
let infoCount = 0
if (!response.ok) {
@@ -219,7 +222,7 @@ export function HealthStatusModal({ open, onOpenChange, getApiUrl }: HealthStatu
if (open) {
fetchHealthDetails()
// Auto-refresh every 5 minutes while modal is open
const refreshInterval = setInterval(fetchHealthDetails, 300000)
const refreshInterval = setInterval(() => fetchHealthDetails(), 300000)
return () => clearInterval(refreshInterval)
}
}, [open, fetchHealthDetails])
@@ -722,6 +725,23 @@ export function HealthStatusModal({ open, onOpenChange, getApiUrl }: HealthStatu
No issues detected
</div>
)}
{/* Only offer "Update Now" when the category is not
already OK — hiding it when there's nothing
pending prevents the operator from spawning a
terminal that would only report "System is
already up to date". */}
{key === "updates" && status?.toUpperCase() !== "OK" && (
<div className="flex justify-end px-3 py-2 pt-1">
<Button
size="sm"
onClick={() => setShowUpdateTerminal(true)}
className="bg-purple-600/15 hover:bg-purple-600/25 border border-purple-500/40 text-purple-300 hover:text-purple-200"
>
<Download className="h-4 w-4 mr-1.5" />
Update Now
</Button>
</div>
)}
</div>
)}
</div>
@@ -848,6 +868,23 @@ export function HealthStatusModal({ open, onOpenChange, getApiUrl }: HealthStatu
</div>
)}
</DialogContent>
<ScriptTerminalModal
open={showUpdateTerminal}
onClose={() => {
setShowUpdateTerminal(false)
// Force a fresh read (cache-busting via ?refresh=1) so the
// "System Updates" row reflects the state right after the
// update finished, instead of the pre-update cached value.
fetchHealthDetails(true).catch(() => {})
}}
scriptPath="/usr/local/share/proxmenux/scripts/utilities/proxmox_update.sh"
scriptName="proxmox_update"
params={{
EXECUTION_MODE: "web",
}}
title="Proxmox System Update"
description="Runs apt-get update + dist-upgrade and post-update cleanup on the host."
/>
</Dialog>
)
}
+280 -34
View File
@@ -250,6 +250,35 @@ function pathKey(path: string[]): string {
return path.join("/")
}
// Trim the visible slider range to a window around the saved +
// recommended values so the track has usable resolution (e.g. CPU
// 60100 instead of the backend's 0100). Derived from stable inputs
// so the range does NOT shift under an active drag.
function computeVisualRange(
values: number[],
backendMin: number,
backendMax: number,
step: number,
): { min: number; max: number } {
const totalRange = Math.max(1, backendMax - backendMin)
// Margin ≈ 25% of total range, clamped to at least 5 steps so tiny
// step sizes (e.g. step=1 on 0100) still get a usable window.
const rawMargin = Math.max(step * 5, Math.round(totalRange * 0.25))
const lo = Math.min(...values)
const hi = Math.max(...values)
const snap = (n: number) => Math.round(n / step) * step
let visMin = Math.max(backendMin, snap(lo - rawMargin))
let visMax = Math.min(backendMax, snap(hi + rawMargin))
// Ensure the window is at least 4 steps wide so the slider has
// room to move even if all inputs collapse to one value.
if (visMax - visMin < step * 4) {
const mid = (visMax + visMin) / 2
visMin = Math.max(backendMin, snap(mid - step * 2))
visMax = Math.min(backendMax, snap(mid + step * 2))
}
return { min: visMin, max: visMax }
}
// ─── Component ───────────────────────────────────────────────────────────────
export function HealthThresholds() {
@@ -394,29 +423,13 @@ export function HealthThresholds() {
}
const renderField = (path: string[], label: string) => {
// Kept for single-value leaves that don't have a warn/crit pair
// (e.g. Memory's swap_critical). The pair-cases route to
// renderThresholdRange below.
const leaf = getLeaf(tree, path)
if (!leaf) return null
const key = pathKey(path)
const editingValue = pending[key] ?? String(leaf.value)
// Visual rules (rebuilt — the original used /40 opacity borders +
// a blue ring stacked on top of the colour border, both of which
// were nearly invisible in read-only mode and stacked weirdly when
// a value was customised):
//
// • Read-only mode (editMode=false): keep severity colour on the
// border at a higher opacity (/70 instead of /40) and on the
// background (/10) so the field is clearly readable, and
// restore foreground colour (no `opacity-70` washout). This is
// the default state the user sees most of the time — it must
// match the visual weight of the rest of the Settings page.
// • Edit mode + value matches the recommended default: severity
// border + soft severity bg, same as read-only.
// • Edit mode + value customised: ONE border in blue, replacing
// (not stacking on top of) the severity border. This is the
// single signal that "this value differs from recommended".
//
// `swap_critical` and any other `*_critical` leaf falls into the
// red bucket via the substring check.
const last = path[path.length - 1] || ""
const isCritical = last.toLowerCase().includes("critical")
const isWarning = last.toLowerCase().includes("warning")
@@ -456,6 +469,206 @@ export function HealthThresholds() {
)
}
// Single-handle slider for thresholds that don't have a warn/crit
// pair (only Memory's swap_critical today). Same visual language as
// the dual-handle: red handle, value above, OK / CRIT zones below
// — so the operator doesn't read it as a different control.
const renderSingleThresholdSlider = (path: string[], severity: "warning" | "critical" = "critical") => {
const leaf = getLeaf(tree, path)
if (!leaf) return null
const key = pathKey(path)
const val = Number(pending[key] ?? leaf.value)
const step = leaf.step || 1
const unit = leaf.unit || ""
const { min, max } = computeVisualRange(
[leaf.value, leaf.recommended],
leaf.min,
leaf.max,
step,
)
const pct = ((Math.max(min, Math.min(max, val)) - min) / (max - min)) * 100
const custom = leaf.customised && !(key in pending)
const color = severity === "critical" ? "red" : "amber"
const handleClass = severity === "critical"
? "[&::-webkit-slider-thumb]:bg-red-500 [&::-moz-range-thumb]:bg-red-500"
: "[&::-webkit-slider-thumb]:bg-amber-500 [&::-moz-range-thumb]:bg-amber-500"
const numberColor = custom
? "text-blue-400"
: severity === "critical"
? "text-red-500"
: "text-amber-500"
const fillColor = severity === "critical" ? "bg-red-500/30" : "bg-amber-500/30"
return (
<div className="px-1 py-3">
<div className="relative h-6 sm:h-5 mb-1 select-none">
<span
className={`absolute -translate-x-1/2 text-xs font-semibold tabular-nums ${numberColor}`}
style={{ left: `${pct}%` }}
>
{val}{unit}
</span>
</div>
<div className="relative h-9 sm:h-6">
<div className="absolute inset-x-0 top-1/2 -translate-y-1/2 h-1.5 rounded-full bg-muted" />
<div
className={`absolute top-1/2 -translate-y-1/2 h-1.5 rounded-r-full ${fillColor}`}
style={{ left: `${pct}%`, right: 0 }}
/>
<input
type="range"
min={min}
max={max}
step={step}
disabled={!editMode}
value={val}
onChange={(e) => setPending((p) => ({ ...p, [key]: e.target.value }))}
className={`absolute inset-0 w-full appearance-none bg-transparent pointer-events-none [&::-webkit-slider-thumb]:pointer-events-auto [&::-moz-range-thumb]:pointer-events-auto [&::-webkit-slider-thumb]:appearance-none [&::-webkit-slider-thumb]:h-8 [&::-webkit-slider-thumb]:w-8 sm:[&::-webkit-slider-thumb]:h-4 sm:[&::-webkit-slider-thumb]:w-4 [&::-webkit-slider-thumb]:rounded-full [&::-webkit-slider-thumb]:border-2 [&::-webkit-slider-thumb]:border-background [&::-webkit-slider-thumb]:shadow [&::-moz-range-thumb]:h-8 [&::-moz-range-thumb]:w-8 sm:[&::-moz-range-thumb]:h-4 sm:[&::-moz-range-thumb]:w-4 [&::-moz-range-thumb]:rounded-full [&::-moz-range-thumb]:border-2 [&::-moz-range-thumb]:border-background ${handleClass}`}
title={`Recommended: ${leaf.recommended}${unit}`}
/>
</div>
<div className="grid grid-cols-2 gap-2 mt-1.5 text-[10px] uppercase tracking-wider text-muted-foreground">
<span>OK &lt; {val}{unit}</span>
<span className="text-right">{severity === "critical" ? "CRIT" : "WARN"} &gt; {val}{unit}</span>
</div>
</div>
)
}
// Dual-handle range slider replacing the two stacked number inputs
// for warn/crit pairs. Visual model: a horizontal track split into
// three zones — OK (left of warning, muted), WARN→CRIT (between
// handles, amber-to-red gradient), and OVER-CRIT (right of critical,
// dark red overlay). The handles themselves stay coloured (amber for
// warning, red for critical) so the operator reads the same severity
// mapping they're used to from the old inputs. Numbers ride above
// each handle and double as a click-to-edit affordance — clicking a
// number swaps it for a tight `<Input type="number">` so precise
// values are still keyboard-friendly. Customised values surface as a
// small blue dot on the affected handle (same signal as the old blue
// ring, less visual weight).
const renderThresholdRange = (
basePath: string[],
options?: { hideLabels?: boolean }
) => {
const wPath = [...basePath, "warning"]
const cPath = [...basePath, "critical"]
const wLeaf = getLeaf(tree, wPath)
const cLeaf = getLeaf(tree, cPath)
if (!wLeaf || !cLeaf) return null
const wKey = pathKey(wPath)
const cKey = pathKey(cPath)
const wVal = Number(pending[wKey] ?? wLeaf.value)
const cVal = Number(pending[cKey] ?? cLeaf.value)
// Backend validates warning <= critical on save.
const step = Math.max(wLeaf.step, cLeaf.step) || 1
const backendMin = Math.min(wLeaf.min, cLeaf.min)
const backendMax = Math.max(wLeaf.max, cLeaf.max)
const { min, max } = computeVisualRange(
[wLeaf.value, cLeaf.value, wLeaf.recommended, cLeaf.recommended],
backendMin,
backendMax,
step,
)
const pct = (v: number) => ((Math.max(min, Math.min(max, v)) - min) / (max - min)) * 100
const wPct = pct(wVal)
const cPct = pct(cVal)
const unit = wLeaf.unit || cLeaf.unit || ""
const wCustom = wLeaf.customised && !(wKey in pending)
const cCustom = cLeaf.customised && !(cKey in pending)
const setVal = (key: string, value: number, peer: number, isWarn: boolean) => {
// Clamp on the fly: warning can't cross critical and vice-versa,
// matching the backend invariant so the user can't drag into an
// invalid state.
let v = value
if (isWarn && v >= peer) v = peer - step
if (!isWarn && v <= peer) v = peer + step
setPending((p) => ({ ...p, [key]: String(v) }))
}
return (
<div className="px-1 py-3">
{/* Numeric labels above each handle, positioned absolutely so
they ride above the corresponding thumb regardless of the
slider width. Pointer events disabled so they don't steal
clicks from the underlying range inputs. */}
<div className="relative h-6 sm:h-5 mb-1 select-none">
<span
className={`absolute -translate-x-1/2 text-xs font-semibold tabular-nums ${wCustom ? "text-blue-400" : "text-amber-500"}`}
style={{ left: `${wPct}%` }}
>
{wVal}{unit}
</span>
<span
className={`absolute -translate-x-1/2 text-xs font-semibold tabular-nums ${cCustom ? "text-blue-400" : "text-red-500"}`}
style={{ left: `${cPct}%` }}
>
{cVal}{unit}
</span>
</div>
{/* Two range inputs stacked. Mobile track box is taller so the
enlarged thumbs (h-7) have room to sit without clipping. */}
<div className="relative h-9 sm:h-6">
{/* Background track: OK zone (muted) running the full width */}
<div className="absolute inset-x-0 top-1/2 -translate-y-1/2 h-1.5 rounded-full bg-muted" />
{/* Warn-to-Crit gradient between the two handles */}
<div
className="absolute top-1/2 -translate-y-1/2 h-1.5 rounded-full"
style={{
left: `${wPct}%`,
width: `${Math.max(0, cPct - wPct)}%`,
background: "linear-gradient(90deg, rgb(245 158 11), rgb(239 68 68))",
}}
/>
{/* OVER-CRIT zone (right of critical) — solid red dim */}
<div
className="absolute top-1/2 -translate-y-1/2 h-1.5 rounded-r-full bg-red-500/30"
style={{ left: `${cPct}%`, right: 0 }}
/>
{/* Two superposed range inputs. Pointer-events on the thumb
only, so the inert track bar above stays visible. */}
<input
type="range"
min={min}
max={max}
step={step}
disabled={!editMode}
value={wVal}
onChange={(e) => setVal(wKey, Number(e.target.value), cVal, true)}
className="absolute inset-0 w-full appearance-none bg-transparent pointer-events-none [&::-webkit-slider-thumb]:pointer-events-auto [&::-moz-range-thumb]:pointer-events-auto [&::-webkit-slider-thumb]:appearance-none [&::-webkit-slider-thumb]:h-8 [&::-webkit-slider-thumb]:w-8 sm:[&::-webkit-slider-thumb]:h-4 sm:[&::-webkit-slider-thumb]:w-4 [&::-webkit-slider-thumb]:rounded-full [&::-webkit-slider-thumb]:bg-amber-500 [&::-webkit-slider-thumb]:border-2 [&::-webkit-slider-thumb]:border-background [&::-webkit-slider-thumb]:shadow [&::-moz-range-thumb]:h-8 [&::-moz-range-thumb]:w-8 sm:[&::-moz-range-thumb]:h-4 sm:[&::-moz-range-thumb]:w-4 [&::-moz-range-thumb]:rounded-full [&::-moz-range-thumb]:bg-amber-500 [&::-moz-range-thumb]:border-2 [&::-moz-range-thumb]:border-background"
title={`Warning (recommended: ${wLeaf.recommended}${unit})`}
/>
<input
type="range"
min={min}
max={max}
step={step}
disabled={!editMode}
value={cVal}
onChange={(e) => setVal(cKey, Number(e.target.value), wVal, false)}
className="absolute inset-0 w-full appearance-none bg-transparent pointer-events-none [&::-webkit-slider-thumb]:pointer-events-auto [&::-moz-range-thumb]:pointer-events-auto [&::-webkit-slider-thumb]:appearance-none [&::-webkit-slider-thumb]:h-8 [&::-webkit-slider-thumb]:w-8 sm:[&::-webkit-slider-thumb]:h-4 sm:[&::-webkit-slider-thumb]:w-4 [&::-webkit-slider-thumb]:rounded-full [&::-webkit-slider-thumb]:bg-red-500 [&::-webkit-slider-thumb]:border-2 [&::-webkit-slider-thumb]:border-background [&::-webkit-slider-thumb]:shadow [&::-moz-range-thumb]:h-8 [&::-moz-range-thumb]:w-8 sm:[&::-moz-range-thumb]:h-4 sm:[&::-moz-range-thumb]:w-4 [&::-moz-range-thumb]:rounded-full [&::-moz-range-thumb]:bg-red-500 [&::-moz-range-thumb]:border-2 [&::-moz-range-thumb]:border-background"
title={`Critical (recommended: ${cLeaf.recommended}${unit})`}
/>
</div>
{/* Zone labels — explicit ranges so the operator knows where
"warn" starts and ends without having to read the handles. */}
{!options?.hideLabels && (
<div className="grid grid-cols-3 gap-2 mt-1.5 text-[10px] uppercase tracking-wider text-muted-foreground">
<span>OK &lt; {wVal}{unit}</span>
<span className="text-center">WARN {wVal}{cVal}{unit}</span>
<span className="text-right">CRIT &gt; {cVal}{unit}</span>
</div>
)}
</div>
)
}
return (
<Card>
<CardHeader>
@@ -518,9 +731,9 @@ export function HealthThresholds() {
</div>
<CardDescription>
The Health Monitor and notifications fire when these thresholds are crossed.
Amber inputs are warning levels, red inputs are critical levels. A blue ring
marks a value you've customised away from the recommended default hover the
field to see the recommendation, or use Reset to restore it.
Drag the amber handle to set the warning level and the red handle to set the
critical level. Values that differ from the recommended default appear in blue
hover a handle to see the recommendation, or use Reset to restore it.
</CardDescription>
</CardHeader>
<CardContent>
@@ -556,7 +769,7 @@ export function HealthThresholds() {
<Icon className="h-4 w-4 text-muted-foreground flex-shrink-0" />
<h4 className="text-sm font-medium">{section.title}</h4>
</div>
{!editMode && (
{editMode && (
<button
className="h-6 w-6 rounded-md text-muted-foreground hover:bg-muted hover:text-foreground transition-colors flex items-center justify-center"
onClick={() => handleResetSection(section.id)}
@@ -571,18 +784,51 @@ export function HealthThresholds() {
{section.description}
</p>
)}
<div className="divide-y divide-border/40">
{section.rowGroups
? section.rowGroups.map((group) => (
<div key={group.subKey} className="py-1.5">
<div className="text-[11px] uppercase tracking-wider text-muted-foreground mb-0.5 px-1">
{group.label}
</div>
{renderField([section.id, group.subKey, "warning"], "Warning")}
{renderField([section.id, group.subKey, "critical"], "Critical")}
<div>
{section.rowGroups ? (
// Per-class disk temperature: one slider per row
// (HDD / SSD / NVMe / SAS). Group label sits on
// top of each slider so the operator scans the
// column from top down without losing context.
section.rowGroups.map((group) => (
<div key={group.subKey} className="py-1.5 border-b border-border/40 last:border-b-0">
<div className="text-[11px] uppercase tracking-wider text-muted-foreground px-1">
{group.label}
</div>
))
: section.fields.map((f) => renderField(f.path, f.label))}
{renderThresholdRange([section.id, group.subKey])}
</div>
))
) : section.id === "memory" ? (
// Memory & Swap is special: warn/crit pair for
// RAM, plus a single Swap threshold that has no
// companion (it's a "critical only" metric).
// Both use sliders so the section reads as one
// visual language end to end.
<>
<div className="text-[11px] uppercase tracking-wider text-muted-foreground px-1">
RAM
</div>
{renderThresholdRange(["memory"])}
<div className="border-t border-border/40">
<div className="text-[11px] uppercase tracking-wider text-muted-foreground px-1 pt-1.5">
Swap (critical only)
</div>
{renderSingleThresholdSlider(["memory", "swap_critical"], "critical")}
</div>
</>
) : section.fields.length === 2 &&
section.fields[0].path[section.fields[0].path.length - 1] === "warning" &&
section.fields[1].path[section.fields[1].path.length - 1] === "critical" ? (
// Generic warn+crit pair (CPU, CPU temp, storage
// capacities …) → single slider.
renderThresholdRange([section.id])
) : (
// Fallback for any future section shape — keep
// the original per-field number inputs.
<div className="divide-y divide-border/40">
{section.fields.map((f) => renderField(f.path, f.label))}
</div>
)}
</div>
</div>
)
File diff suppressed because it is too large Load Diff
+1 -1
View File
@@ -271,7 +271,7 @@ export function Login({ onLogin }: LoginProps) {
</form>
</div>
<p className="text-center text-sm text-muted-foreground">ProxMenux Monitor v1.2.2.1-beta</p>
<p className="text-center text-sm text-muted-foreground">ProxMenux Monitor v1.2.4</p>
</div>
</div>
)
File diff suppressed because it is too large Load Diff
+461 -140
View File
@@ -4,9 +4,10 @@ import { useEffect, useState } from "react"
import { Card, CardContent, CardHeader, CardTitle } from "./ui/card"
import { Badge } from "./ui/badge"
import { Dialog, DialogContent, DialogHeader, DialogTitle, DialogDescription } from "./ui/dialog"
import { Wifi, Activity, Network, Router, AlertCircle, Zap, Timer } from 'lucide-react'
import { Wifi, Activity, Network, Router, AlertCircle, Zap, Timer, EthernetPort, ArrowDown, ArrowUp, Box, ChevronRight } from 'lucide-react'
import useSWR from "swr"
import { NetworkTrafficChart } from "./network-traffic-chart"
import { NetworkFlow, type NetworkFlowData } from "./network-flow"
import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "./ui/select"
import { fetchApi } from "../lib/api-config"
import { formatNetworkTraffic, getNetworkUnit } from "../lib/format-network"
@@ -17,6 +18,9 @@ interface NetworkData {
interfaces: NetworkInterface[]
physical_interfaces?: NetworkInterface[]
bridge_interfaces?: NetworkInterface[]
// Bond masters. Also present inside `interfaces` for backward
// compatibility; this list is what the topology diagram consumes.
bond_interfaces?: NetworkInterface[]
vm_lxc_interfaces?: NetworkInterface[]
traffic: {
bytes_sent: number
@@ -63,12 +67,41 @@ interface NetworkInterface {
errors_out?: number
drops_in?: number
drops_out?: number
// Live rate (bytes/sec) computed by the backend as the delta
// between this poll and the previous one. Present from the second
// /api/network response onward; absent on the first call after the
// service starts or after a long pause.
rx_Bps?: number
tx_Bps?: number
// Hardware ceiling parsed from ethtool's "Supported link modes".
// The card shows "(max N Gbps)" next to the negotiated speed when
// the link is auto-negotiated below the NIC's max.
max_speed?: number
// Bridges that have this physical NIC as their underlying interface
// (directly, or as a bond slave). Surfaced in the card so the
// operator can see "this NIC → vmbr0" at a glance.
used_by_bridges?: string[]
bond_mode?: string
// Kernel's human-readable mode, e.g. "fault-tolerance (active-backup)".
// bond_mode holds the short form ("active-backup") that matches
// /etc/network/interfaces and the Proxmox UI.
bond_mode_detail?: string | null
bond_slaves?: string[]
bond_active_slave?: string | null
// True only for modes where a slave really sits idle (active-backup).
bond_supports_failover?: boolean
bond_slave_status?: Record<string, string>
// Set on a physical NIC that is enslaved to a bond.
bond_master?: string
bond_role?: "active" | "standby" | "member"
bond_link?: string
// Master device resolved from /sys/class/net/<iface>/master — the
// bridge for a guest tap, the bond for a slave NIC.
bridge_owner?: string
bridge_members?: string[]
bridge_physical_interface?: string
bridge_bond_slaves?: string[]
bridge_vlan_interface?: string | null
packet_loss_in?: number
packet_loss_out?: number
vmid?: number
@@ -77,6 +110,231 @@ interface NetworkInterface {
vm_status?: string
}
// Same dot-prefix tone the Storage cards use, so a "no errors" /
// "errors present" cue reads identically across pages.
const NetStatusDot = ({ tone }: { tone: "ok" | "warn" | "fail" }) => {
const cls =
tone === "ok" ? "bg-green-500" : tone === "warn" ? "bg-yellow-500" : "bg-red-500"
return <span className={`inline-block h-2 w-2 rounded-full shrink-0 ${cls}`} aria-hidden />
}
const netCounterTone = (n: number | null | undefined): "ok" | "warn" | "fail" => {
if (!n || n <= 0) return "ok"
if (n < 10) return "warn"
return "fail"
}
// Icon picker — defaults to the actual port type rather than a Wi-Fi
// glyph for everything. Wireless interfaces (wl*/wifi*) keep the Wi-Fi
// glyph; wired NICs use EthernetPort; bonds/bridges/vlans get more
// specific icons so the operator can tell them apart at a glance.
function getInterfaceIcon(iface: NetworkInterface): React.ComponentType<{ className?: string }> {
const name = (iface.name || "").toLowerCase()
const type = (iface.type || "").toLowerCase()
if (name.startsWith("wl") || name.startsWith("wifi")) return Wifi
if (type === "bridge") return Network
if (type === "bond") return Router
if (type === "vlan") return Activity
if (type === "vm_lxc" || type === "virtual") return Box
// Physical wired NIC (eth0, enp*, ens*, eno*, nic0, …) → ethernet port.
return EthernetPort
}
// Match the dark blue badge tone the Storage card uses for the disk
// type chip, but mapped to the actual interface class.
function getInterfaceTypeChip(type: string) {
switch ((type || "").toLowerCase()) {
case "physical":
return { className: "bg-blue-500/10 text-blue-400 border-blue-500/20", label: "Physical" }
case "bridge":
return { className: "bg-green-500/10 text-green-400 border-green-500/20", label: "Bridge" }
case "bond":
return { className: "bg-purple-500/10 text-purple-400 border-purple-500/20", label: "Bond" }
case "vlan":
return { className: "bg-cyan-500/10 text-cyan-400 border-cyan-500/20", label: "VLAN" }
case "vm_lxc":
case "virtual":
return { className: "bg-orange-500/10 text-orange-400 border-orange-500/20", label: "Virtual" }
default:
return { className: "bg-gray-500/10 text-gray-400 border-gray-500/20", label: type || "Unknown" }
}
}
// Per-interface card matching the Storage page's "Physical Disks"
// pattern: 2-line header (identity / live state), horizontal divider,
// vertical key→value stat block, footer with serial + arrow CTA.
// Replaces the row-style block that was unchanged since 1.0.0.
function renderPhysicalInterfaceCardV2(
iface: NetworkInterface,
onOpen: (iface: NetworkInterface) => void,
) {
const Icon = getInterfaceIcon(iface)
const chip = getInterfaceTypeChip(iface.type)
const isUp = (iface.status || "").toLowerCase() === "up"
const firstAddr = iface.addresses?.[0]?.ip || ""
const extraAddrs = Math.max(0, (iface.addresses?.length || 0) - 1)
const speedStr = formatSpeed(iface.speed)
// Hardware max in Mbps from ethtool. Show only when it's different
// from the negotiated speed (avoids "1 Gbps (max 1 Gbps)" noise).
const maxSpeedStr =
iface.max_speed && iface.max_speed !== iface.speed
? formatSpeed(iface.max_speed)
: ""
const bridgesUsing = iface.used_by_bridges || []
const errIn = iface.errors_in ?? 0
const errOut = iface.errors_out ?? 0
const dropIn = iface.drops_in ?? 0
const dropOut = iface.drops_out ?? 0
const totalErrors = errIn + errOut
const totalDrops = dropIn + dropOut
return (
<div
key={iface.name}
className="border border-white/10 rounded-lg p-5 cursor-pointer bg-card hover:bg-white/5 transition-colors flex flex-col"
onClick={() => onOpen(iface)}
>
{/* Header L1: identity (icon + name + type) | status. */}
<div className="flex items-start justify-between gap-3">
<div className="flex items-center gap-2 flex-wrap min-w-0">
<Icon className="h-5 w-5 text-muted-foreground shrink-0" />
<h3 className="font-mono font-bold text-base break-all">{iface.name}</h3>
<Badge variant="outline" className={chip.className}>{chip.label}</Badge>
</div>
<span
className={`flex items-center gap-1.5 text-sm font-semibold uppercase tracking-wide shrink-0 ${
isUp ? "text-green-500" : "text-red-400"
}`}
>
<NetStatusDot tone={isUp ? "ok" : "fail"} />
{iface.status || "?"}
</span>
</div>
{/* Header L2: speed + max (when negotiated < hw) | duplex. */}
<div className="flex items-center justify-between gap-3 mt-1 text-sm text-muted-foreground">
<span className="flex items-center gap-1.5">
<Zap className="h-3.5 w-3.5" />
{speedStr}
{maxSpeedStr && (
<span className="text-[11px] text-muted-foreground/70">
· max {maxSpeedStr}
</span>
)}
</span>
<span className="capitalize">{iface.duplex || "—"}</span>
</div>
{/* Separator. */}
<div className="border-t border-border/60 my-3" />
{/* Stats: key uppercase left · value right. */}
<div className="space-y-2 text-sm">
{firstAddr && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground shrink-0">
IP
</span>
<span className="font-medium text-right truncate font-mono text-xs">
{firstAddr}{extraAddrs > 0 ? ` (+${extraAddrs})` : ""}
</span>
</div>
)}
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground">MTU</span>
<span className="font-medium">{iface.mtu || "—"}</span>
</div>
{bridgesUsing.length > 0 && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground shrink-0">
Bridge
</span>
<span className="font-medium text-right truncate font-mono text-xs text-cyan-400">
{bridgesUsing.map((b) => `${b}`).join(" ")}
</span>
</div>
)}
{/* Live RX/TX rate. Same wording the Network Traffic chart
uses ("Received" / "Sent") and the same canonical colours
(green for Received, blue for Sent). Falls back to "—"
until the backend has a delta — first poll after start
has no previous sample to compute against. */}
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground flex items-center gap-1">
<ArrowDown className="h-3 w-3 text-green-500" /> Received
</span>
<span className="font-medium text-green-500 tabular-nums">
{iface.rx_Bps !== undefined ? formatRate(iface.rx_Bps) : "—"}
</span>
</div>
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground flex items-center gap-1">
<ArrowUp className="h-3 w-3 text-blue-400" /> Sent
</span>
<span className="font-medium text-blue-400 tabular-nums">
{iface.tx_Bps !== undefined ? formatRate(iface.tx_Bps) : "—"}
</span>
</div>
{(totalErrors > 0 || totalDrops > 0) && (
<>
{totalErrors > 0 && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground">Errors</span>
<span
className={`font-medium flex items-center gap-1.5 ${
netCounterTone(totalErrors) === "ok"
? "text-green-500"
: netCounterTone(totalErrors) === "warn"
? "text-yellow-500"
: "text-red-500"
}`}
>
<NetStatusDot tone={netCounterTone(totalErrors)} />
{totalErrors.toLocaleString()}
</span>
</div>
)}
{totalDrops > 0 && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground">Drops</span>
<span
className={`font-medium flex items-center gap-1.5 ${
netCounterTone(totalDrops) === "ok"
? "text-green-500"
: netCounterTone(totalDrops) === "warn"
? "text-yellow-500"
: "text-red-500"
}`}
>
<NetStatusDot tone={netCounterTone(totalDrops)} />
{totalDrops.toLocaleString()}
</span>
</div>
)}
</>
)}
</div>
{/* Footer: MAC (left, mono) + arrow CTA (right). */}
<div className="border-t border-border/60 mt-auto pt-3 flex items-center justify-between gap-3">
{iface.mac_address ? (
<span className="text-[11px] text-foreground font-mono truncate min-w-0">
<span className="text-muted-foreground">MAC:</span> {iface.mac_address}
</span>
) : (
<span />
)}
<span
className="text-blue-400 hover:text-blue-300 transition-colors text-base leading-none shrink-0"
aria-label="View details"
>
</span>
</div>
</div>
)
}
const getInterfaceTypeBadge = (type: string) => {
switch (type) {
case "physical":
@@ -105,6 +363,19 @@ const getVMTypeBadge = (vmType: string | undefined) => {
return { color: "bg-gray-500/10 text-gray-500 border-gray-500/20", label: "Unknown" }
}
// Format bytes/sec into the canonical network unit ladder.
// Matches the convention used by the Network Traffic chart so the
// rates on the per-interface cards and the chart read the same way.
const formatRate = (bps: number | undefined): string => {
if (bps === undefined || bps === null || !Number.isFinite(bps)) return "—"
if (bps < 1) return "0 B/s"
const k = 1024
const sizes = ["B/s", "KB/s", "MB/s", "GB/s"]
const i = Math.min(sizes.length - 1, Math.floor(Math.log(bps) / Math.log(k)))
const v = bps / Math.pow(k, i)
return `${v >= 100 ? v.toFixed(0) : v.toFixed(v >= 10 ? 1 : 2)} ${sizes[i]}`
}
const formatBytes = (bytes: number | undefined): string => {
if (!bytes || bytes === 0) return "0 B"
const k = 1024
@@ -142,7 +413,10 @@ export function NetworkMetrics() {
error,
isLoading,
} = useSWR<NetworkData>("/api/network", fetcher, {
refreshInterval: 15000,
// Was 15 s — too long for the Network Flow's pulse animation
// which needs near-live rates. 3 s gives the dashboard responsive
// updates without hammering the backend.
refreshInterval: 3000,
revalidateOnFocus: true,
revalidateOnReconnect: true,
})
@@ -253,6 +527,7 @@ export function NetworkMetrics() {
const allInterfaces = [
...(networkData.physical_interfaces || []),
...(networkData.bond_interfaces || []),
...(networkData.bridge_interfaces || []),
...(networkData.vm_lxc_interfaces || []),
]
@@ -291,6 +566,25 @@ export function NetworkMetrics() {
}
}
// Compact form for inline header use. The full "24 Hours" gets noisy
// next to the title; "Past 24 h" keeps the same meaning in less space.
const getTimeframeShortLabel = () => {
switch (timeframe) {
case "hour":
return "Past 1 h"
case "day":
return "Past 24 h"
case "week":
return "Past 7 d"
case "month":
return "Past 30 d"
case "year":
return "Past 1 y"
default:
return "Past 24 h"
}
}
const hostname = networkData.hostname || "N/A"
const domain = networkData.domain || "N/A"
const dnsServers = networkData.dns_servers || []
@@ -300,7 +594,7 @@ export function NetworkMetrics() {
return (
<div className="space-y-6">
{/* Network Overview Cards */}
<div className="grid grid-cols-1 sm:grid-cols-2 lg:grid-cols-4 gap-3 lg:gap-6">
<div className="grid grid-cols-1 sm:grid-cols-2 xl:grid-cols-4 gap-3 xl:gap-6">
{/* ── Network Traffic (preview restyle: Down/Up dual headline + stacked bar) ── */}
{(() => {
const downBytes = networkData.traffic.bytes_recv || 0
@@ -311,8 +605,11 @@ export function NetworkMetrics() {
return (
<Card className="bg-card border-border">
<CardHeader className="flex flex-row items-center justify-between space-y-0 pb-2">
<CardTitle className="text-sm font-medium text-muted-foreground">Network Traffic</CardTitle>
<Activity className="h-4 w-4 text-muted-foreground" />
<div className="flex flex-col gap-0.5 min-w-0">
<CardTitle className="text-sm font-medium text-muted-foreground">Network Traffic</CardTitle>
<span className="text-[10px] text-muted-foreground/70 font-normal">{getTimeframeShortLabel()}</span>
</div>
<Activity className="h-4 w-4 text-muted-foreground flex-shrink-0" />
</CardHeader>
<CardContent>
<div className="grid grid-cols-2 gap-3 mb-3">
@@ -409,13 +706,16 @@ export function NetworkMetrics() {
</Card>
{/* Latency Card with Sparkline */}
<Card
className="bg-card border-border cursor-pointer hover:bg-muted/50 transition-colors"
<Card
className="bg-card border-border cursor-pointer hover:bg-white/5 transition-colors"
onClick={() => setLatencyModalOpen(true)}
>
<CardHeader className="flex flex-row items-center justify-between space-y-0 pb-2">
<CardTitle className="text-sm font-medium text-muted-foreground">Network Latency</CardTitle>
<Timer className="h-4 w-4 text-muted-foreground" />
<div className="flex items-center gap-1 text-muted-foreground">
<Timer className="h-4 w-4" />
<ChevronRight className="h-4 w-4 opacity-60" />
</div>
</CardHeader>
<CardContent>
<div className="flex items-center justify-between mb-2">
@@ -500,6 +800,105 @@ export function NetworkMetrics() {
</CardContent>
</Card>
{/* Network Flow — proof of concept. Lives next to Network Traffic
while the design is validated on a real host. Once approved,
this card replaces (or pairs with) the chart above. */}
{(() => {
const toMBps = (bps?: number) => (bps || 0) / (1024 * 1024)
const allIfaces = [
...(networkData.physical_interfaces || []),
...(networkData.bond_interfaces || []),
...(networkData.bridge_interfaces || []),
...(networkData.vm_lxc_interfaces || []),
]
const flowData: NetworkFlowData = {
nics: (networkData.physical_interfaces || []).map((p) => ({
id: p.name,
link: formatSpeed(p.speed),
rx: toMBps(p.rx_Bps),
tx: toMBps(p.tx_Bps),
// A slave whose MII status is down has no carrier even though
// the interface itself stays administratively up — the bond
// driver's view is the accurate one here.
status:
p.bond_link === "down"
? "down"
: (p.status || "").toLowerCase() === "up"
? "up"
: "down",
bond: p.bond_master,
role: p.bond_role,
})),
bonds: (networkData.bond_interfaces || []).map((b) => ({
id: b.name,
mode: b.bond_mode && b.bond_mode !== "unknown" ? b.bond_mode : undefined,
rx: toMBps(b.rx_Bps),
tx: toMBps(b.tx_Bps),
status: (b.status || "").toLowerCase() === "up" ? "up" : "down",
})),
bridges: (networkData.bridge_interfaces || []).map((b) => ({
id: b.name,
parent: b.bridge_physical_interface,
})),
consumers: [
(() => {
// PROXMOX node = sum of every running guest's rate.
// This stays consistent with each bridge's own label
// (which sums the same guest rates), and with the
// total trunk flow — no discrepancy between the host's
// displayed rate and the sum of its bridges.
const runningGuests = (networkData.vm_lxc_interfaces || []).filter(
(v) => v.vm_status !== "stopped"
)
return {
id: "host",
label: "host",
kind: "host" as const,
bridge: (networkData.bridge_interfaces?.[0]?.name) || "",
rx: runningGuests.reduce((a, v) => a + toMBps(v.rx_Bps), 0),
tx: runningGuests.reduce((a, v) => a + toMBps(v.tx_Bps), 0),
}
})(),
...(networkData.vm_lxc_interfaces || []).map((v) => {
// Authoritative bridge from the kernel (read by the
// backend from /sys/class/net/<iface>/master). Fallback
// to bridge_members scan, then first bridge as last
// resort so we never silently drop a guest.
const ownerName =
(v as any).bridge_owner ||
(networkData.bridge_interfaces || []).find((b) =>
(b.bridge_members || []).includes(v.name)
)?.name ||
(networkData.bridge_interfaces?.[0]?.name || "")
return {
id: v.name,
label: v.vm_name || v.name,
kind: (v.vm_type === "vm" ? "vm" : "lxc") as "vm" | "lxc",
bridge: ownerName,
rx: toMBps(v.rx_Bps),
tx: toMBps(v.tx_Bps),
offline: v.vm_status === "stopped",
}
}),
],
}
return (
<NetworkFlow
data={flowData}
onNodeClick={(name) => {
// Map the clicked node back to a NetworkInterface and
// open the same details modal the cards below use. The
// virtual "host" id never matches a real interface, so
// it's a no-op — tapping the PROXMOX circle does nothing
// (there's no host-level modal in this view).
if (name === "host") return
const match = allIfaces.find((iface) => iface.name === name)
if (match) setSelectedInterface(match)
}}
/>
)
})()}
{/* Physical Interfaces section */}
<Card className="bg-card border-border">
<CardHeader>
@@ -512,76 +911,13 @@ export function NetworkMetrics() {
</CardTitle>
</CardHeader>
<CardContent>
<div className="space-y-4">
{networkData.physical_interfaces.map((interface_, index) => {
const typeBadge = getInterfaceTypeBadge(interface_.type)
return (
<div
key={index}
className="flex flex-col gap-3 p-4 rounded-lg border border-white/10 bg-white/5 sm:bg-card sm:hover:bg-white/5 transition-colors cursor-pointer"
onClick={() => setSelectedInterface(interface_)}
>
{/* First row: Icon, Name, Type Badge, Status */}
<div className="flex items-center gap-3 flex-wrap">
<Wifi className="h-5 w-5 text-muted-foreground flex-shrink-0" />
<div className="flex items-center gap-2 min-w-0 flex-1 flex-wrap">
<div className="font-medium text-foreground">{interface_.name}</div>
<Badge variant="outline" className={typeBadge.color}>
{typeBadge.label}
</Badge>
</div>
<Badge
variant="outline"
className={
interface_.status === "up"
? "bg-green-500/10 text-green-500 border-green-500/20"
: "bg-red-500/10 text-red-500 border-red-500/20"
}
>
{interface_.status.toUpperCase()}
</Badge>
</div>
{/* Second row: Details - Responsive layout */}
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 text-sm">
<div>
<div className="text-muted-foreground text-xs">IP Address</div>
<div className="font-medium text-foreground font-mono text-sm truncate">
{interface_.addresses.length > 0 ? interface_.addresses[0].ip : "N/A"}
</div>
</div>
<div>
<div className="text-muted-foreground text-xs">Speed</div>
<div className="font-medium text-foreground flex items-center gap-1 text-xs">
<Zap className="h-3 w-3" />
{formatSpeed(interface_.speed)}
</div>
</div>
<div>
<div className="text-muted-foreground text-xs">Duplex</div>
<div className="font-medium text-foreground text-xs capitalize">{interface_.duplex}</div>
</div>
<div>
<div className="text-muted-foreground text-xs">MTU</div>
<div className="font-medium text-foreground text-xs">{interface_.mtu}</div>
</div>
{interface_.mac_address && (
<div className="col-span-2 md:col-span-4">
<div className="text-muted-foreground text-xs">MAC</div>
<div className="font-medium text-foreground font-mono text-xs truncate">
{interface_.mac_address}
</div>
</div>
)}
</div>
</div>
)
})}
{/* Same responsive grid as the Storage page: 3 cols desktop,
2 cols tablet, 1 col mobile. Cards self-size so a row of
long interface names won't push others off-screen. */}
<div className="grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4">
{networkData.physical_interfaces.map((iface) =>
renderPhysicalInterfaceCardV2(iface, setSelectedInterface),
)}
</div>
</CardContent>
</Card>
@@ -610,7 +946,7 @@ export function NetworkMetrics() {
>
{/* First row: Icon, Name, Type Badge, Physical Interface (responsive), Status */}
<div className="flex items-center gap-3 flex-wrap">
<Wifi className="h-5 w-5 text-muted-foreground flex-shrink-0" />
<Network className="h-5 w-5 text-muted-foreground flex-shrink-0" />
<div className="flex items-center gap-2 min-w-0 flex-1 flex-wrap">
<div className="font-medium text-foreground">{interface_.name}</div>
<Badge variant="outline" className={typeBadge.color}>
@@ -619,24 +955,6 @@ export function NetworkMetrics() {
{interface_.bridge_physical_interface && (
<div className="text-sm text-blue-500 font-medium flex items-center gap-1 flex-wrap break-all">
{interface_.bridge_physical_interface}
{interface_.bridge_physical_interface.startsWith("bond") &&
networkData.physical_interfaces && (
<>
{(() => {
const bondInterface = networkData.physical_interfaces.find(
(iface) => iface.name === interface_.bridge_physical_interface,
)
if (bondInterface?.bond_slaves && bondInterface.bond_slaves.length > 0) {
return (
<span className="text-muted-foreground text-xs break-all">
({bondInterface.bond_slaves.join(", ")})
</span>
)
}
return null
})()}
</>
)}
{interface_.bridge_bond_slaves && interface_.bridge_bond_slaves.length > 0 && (
<span className="text-muted-foreground text-xs break-all">
({interface_.bridge_bond_slaves.join(", ")})
@@ -726,14 +1044,14 @@ export function NetworkMetrics() {
>
{/* First row: Icon, Name, VM/LXC Badge, VM Name, Status */}
<div className="flex items-center gap-3 flex-wrap">
<Wifi className="h-5 w-5 text-muted-foreground flex-shrink-0" />
<EthernetPort className="h-5 w-5 text-muted-foreground flex-shrink-0" />
<div className="flex items-center gap-2 min-w-0 flex-1 flex-wrap">
<div className="font-medium text-foreground">{interface_.name}</div>
<Badge variant="outline" className={vmTypeBadge.color}>
{vmTypeBadge.label}
</Badge>
{interface_.vm_name && (
<div className="text-sm text-muted-foreground truncate"> {interface_.vm_name}</div>
<div className="text-sm text-orange-500 truncate"> {interface_.vm_name}</div>
)}
</div>
<Badge
@@ -792,7 +1110,7 @@ export function NetworkMetrics() {
{/* Interface Details Modal */}
<Dialog open={!!selectedInterface} onOpenChange={() => setSelectedInterface(null)}>
<DialogContent className="max-w-4xl max-h-[90vh] overflow-y-auto">
<DialogContent className="max-w-4xl w-[calc(100vw-1rem)] sm:w-[95vw] max-h-[calc(100dvh-2rem)] sm:max-h-[90vh] overflow-y-auto overflow-x-hidden p-4 sm:p-6">
<DialogHeader>
<DialogTitle className="flex items-center gap-2">
<Router className="h-5 w-5" />
@@ -826,6 +1144,7 @@ export function NetworkMetrics() {
const currentInterfaceData = modalNetworkData
? [
...(modalNetworkData.physical_interfaces || []),
...(modalNetworkData.bond_interfaces || []),
...(modalNetworkData.bridge_interfaces || []),
...(modalNetworkData.vm_lxc_interfaces || []),
].find((iface) => iface.name === selectedInterface.name)
@@ -855,35 +1174,10 @@ export function NetworkMetrics() {
<div className="font-medium text-blue-500 text-lg break-all">
{displayInterface.bridge_physical_interface}
</div>
{displayInterface.bridge_physical_interface.startsWith("bond") &&
modalNetworkData?.physical_interfaces && (
<>
{(() => {
const bondInterface = modalNetworkData.physical_interfaces.find(
(iface) => iface.name === displayInterface.bridge_physical_interface,
)
if (bondInterface?.bond_slaves && bondInterface.bond_slaves.length > 0) {
return (
<div className="mt-2">
<div className="text-sm text-muted-foreground mb-2">Bond Members</div>
<div className="flex flex-wrap gap-2">
{bondInterface.bond_slaves.map((slave, idx) => (
<Badge
key={idx}
variant="outline"
className="bg-purple-500/10 text-purple-500 border-purple-500/20"
>
{slave}
</Badge>
))}
</div>
</div>
)
}
return null
})()}
</>
)}
{/* Slaves come from the bridge's own payload
(bridge_bond_slaves); the bond master is not
part of physical_interfaces, so looking it up
there never matched. */}
{displayInterface.bridge_bond_slaves && displayInterface.bridge_bond_slaves.length > 0 && (
<div className="mt-2">
<div className="text-sm text-muted-foreground mb-2">Bond Members</div>
@@ -1119,26 +1413,53 @@ export function NetworkMetrics() {
<div className="space-y-3">
<div>
<div className="text-sm text-muted-foreground">Bonding Mode</div>
<div className="font-medium">{displayInterface.bond_mode || "Unknown"}</div>
<div className="font-medium">
{displayInterface.bond_mode || "Unknown"}
{displayInterface.bond_mode_detail &&
displayInterface.bond_mode_detail !== displayInterface.bond_mode && (
<span className="text-muted-foreground font-normal">
{" "}
({displayInterface.bond_mode_detail})
</span>
)}
</div>
</div>
{displayInterface.bond_active_slave && (
<div>
<div className="text-sm text-muted-foreground">Active Slave</div>
<div className="text-sm text-muted-foreground">
{displayInterface.bond_supports_failover ? "Active Slave" : "Primary Slave"}
</div>
<div className="font-medium">{displayInterface.bond_active_slave}</div>
</div>
)}
<div>
<div className="text-sm text-muted-foreground mb-2">Slave Interfaces</div>
<div className="flex flex-wrap gap-2">
{displayInterface.bond_slaves.map((slave, idx) => (
<Badge
key={idx}
variant="outline"
className="bg-purple-500/10 text-purple-500 border-purple-500/20"
>
{slave}
</Badge>
))}
{displayInterface.bond_slaves.map((slave, idx) => {
// Only active-backup has a real standby. In every
// other mode all slaves transmit, so we just show
// the link state.
const link = displayInterface.bond_slave_status?.[slave]
const isDown = link === "down"
const role = isDown
? "down"
: displayInterface.bond_supports_failover
? slave === displayInterface.bond_active_slave
? "active"
: "standby"
: null
const tone = isDown
? "bg-red-500/10 text-red-500 border-red-500/20"
: role === "standby"
? "bg-yellow-500/10 text-yellow-500 border-yellow-500/20"
: "bg-purple-500/10 text-purple-500 border-purple-500/20"
return (
<Badge key={idx} variant="outline" className={tone}>
{slave}
{role && <span className="ml-1 opacity-70">· {role}</span>}
</Badge>
)
})}
</div>
</div>
</div>
+111 -28
View File
@@ -66,11 +66,68 @@ const CustomMemoryTooltip = ({ active, payload, label }: any) => {
return null
}
interface MetricsError {
headline: string
details?: string
suggestion?: string
}
// AVG / MAX / MIN chip row for the chart card headers. Values come
// from the backend `period_stats` (calculated over the raw RRD points
// BEFORE downsampling), not from the displayed chart points — that's
// what makes a 1-minute CPU spike still appear in the 24h MAX even
// though the chart shows 5-min bucket averages.
//
// Colour choice: all three values render in the same foreground tone.
// The previous red(max)/green(min) scheme misread as severity (a
// healthy 10 % CPU max showed in red and looked like an alert).
//
// Responsive: on ≥sm the chips sit to the right of the title; on
// mobile they wrap below in their own row (the parent CardHeader uses
// `flex-col sm:flex-row`). Smaller text + tabular-nums keeps the
// chips compact enough that they don't crowd long titles.
type PeriodStat = { avg: number; max: number; min: number } | null
function ChartStatsHeader({
stats,
suffix = "",
}: {
stats: PeriodStat
suffix?: string
}) {
if (!stats) return null
const fmt = (n: number) => (n >= 100 ? n.toFixed(0) : n.toFixed(1))
return (
<div className="flex flex-wrap items-baseline gap-x-3 gap-y-1 text-sm tabular-nums">
<span>
<span className="font-semibold text-foreground">{fmt(stats.avg)}{suffix}</span>
<span className="ml-1 text-xs uppercase tracking-wide text-muted-foreground">avg</span>
</span>
<span>
<span className="font-semibold text-foreground">{fmt(stats.max)}{suffix}</span>
<span className="ml-1 text-xs uppercase tracking-wide text-muted-foreground">max</span>
</span>
<span>
<span className="font-semibold text-foreground">{fmt(stats.min)}{suffix}</span>
<span className="ml-1 text-xs uppercase tracking-wide text-muted-foreground">min</span>
</span>
</div>
)
}
export function NodeMetricsCharts() {
const [timeframe, setTimeframe] = useState("day")
const [data, setData] = useState<NodeMetricsData[]>([])
// period_stats from the backend — computed over the raw RRD points
// BEFORE the 5-min downsampling so the chart header's MAX/MIN
// captures real per-minute extremes (a 1-min CPU spike still shows
// up on the 24h view's MAX).
const [periodStats, setPeriodStats] = useState<{
cpu?: PeriodStat
memory_used?: PeriodStat
}>({})
const [loading, setLoading] = useState(true)
const [error, setError] = useState<string | null>(null)
const [error, setError] = useState<MetricsError | null>(null)
const isMobile = useIsMobile()
const [visibleLines, setVisibleLines] = useState({
@@ -158,11 +215,19 @@ export function NodeMetricsCharts() {
})
setData(transformedData)
setPeriodStats(result.period_stats || {})
} catch (err: any) {
console.error("Error fetching node metrics:", err)
console.error("Error message:", err.message)
console.error("Error stack:", err.stack)
setError(err.message || "Error loading metrics")
// fetchApi attaches the parsed JSON body to err.body. The metrics
// endpoint enriches 503 responses with `details` (Proxmox-side
// diagnostic) and `suggestion` (how to fix). Pull them through so
// the user sees actionable text instead of a bare "503".
const body = err?.body
setError({
headline: body?.error || err?.message || "Error loading metrics",
details: body?.details,
suggestion: body?.suggestion,
})
} finally {
setLoading(false)
}
@@ -231,24 +296,36 @@ export function NodeMetricsCharts() {
}
if (error) {
// Both panels carry the same error — render an identical card on
// each side. The headline is the short cause, the details block
// explains it's a Proxmox-host issue (not a Monitor bug), and the
// suggestion is the exact command the operator should run.
const errorCard = (
<Card className="bg-card border-border">
<CardContent className="p-6">
<div className="flex flex-col items-start justify-center h-[300px] gap-2 px-2 overflow-auto">
<p className="text-sm font-semibold text-red-400">{error.headline}</p>
{error.details && (
<p className="text-xs text-muted-foreground leading-relaxed">{error.details}</p>
)}
{error.suggestion && (
<div className="w-full mt-2">
<p className="text-[10px] uppercase tracking-wide text-muted-foreground mb-1">
Suggested fix on the Proxmox host
</p>
<code className="block text-xs bg-background/60 border border-border rounded px-2 py-1.5 font-mono break-all">
{error.suggestion}
</code>
</div>
)}
</div>
</CardContent>
</Card>
)
return (
<div className="grid grid-cols-1 lg:grid-cols-2 gap-6">
<Card className="bg-card border-border">
<CardContent className="p-6">
<div className="flex flex-col items-center justify-center h-[300px] gap-2">
<p className="text-muted-foreground text-sm">Metrics data not available yet</p>
<p className="text-xs text-red-500">{error}</p>
</div>
</CardContent>
</Card>
<Card className="bg-card border-border">
<CardContent className="p-6">
<div className="flex flex-col items-center justify-center h-[300px] gap-2">
<p className="text-muted-foreground text-sm">Metrics data not available yet</p>
<p className="text-xs text-red-500">{error}</p>
</div>
</CardContent>
</Card>
{errorCard}
{errorCard}
</div>
)
}
@@ -298,10 +375,13 @@ export function NodeMetricsCharts() {
{/* CPU Usage + Load Average Chart */}
<Card className="bg-card border-border">
<CardHeader className="px-4 md:px-6">
<CardTitle className="text-foreground flex items-center">
<TrendingUp className="h-5 w-5 mr-2" />
CPU Usage & Load Average
</CardTitle>
<div className="flex flex-col sm:flex-row sm:items-center sm:justify-between gap-2">
<CardTitle className="text-foreground flex items-center">
<TrendingUp className="h-5 w-5 mr-2" />
CPU Usage & Load Average
</CardTitle>
<ChartStatsHeader stats={periodStats.cpu ?? null} suffix="%" />
</div>
</CardHeader>
<CardContent className="px-0 md:px-6">
<ResponsiveContainer width="100%" height={300}>
@@ -370,10 +450,13 @@ export function NodeMetricsCharts() {
{/* Memory Usage Chart */}
<Card className="bg-card border-border">
<CardHeader className="px-4 md:px-6">
<CardTitle className="text-foreground flex items-center">
<MemoryStick className="h-5 w-5 mr-2" />
Memory Usage
</CardTitle>
<div className="flex flex-col sm:flex-row sm:items-center sm:justify-between gap-2">
<CardTitle className="text-foreground flex items-center">
<MemoryStick className="h-5 w-5 mr-2" />
Memory Usage
</CardTitle>
<ChartStatsHeader stats={periodStats.memory_used ?? null} suffix=" GB" />
</div>
</CardHeader>
<CardContent className="px-0 pr-2 md:px-6">
<ResponsiveContainer width="100%" height={300}>
+99 -15
View File
@@ -335,6 +335,12 @@ export function NotificationSettings() {
const [showHistory, setShowHistory] = useState(false)
const [showAdvanced, setShowAdvanced] = useState(false)
const [showSecrets, setShowSecrets] = useState<Record<string, boolean>>({})
// Cleartext secrets cached only while the eye toggle is "on" for
// that field. Settings GET returns "************" for everything in
// SENSITIVE_KEYS; clicking eye fetches the real value via
// /api/notifications/reveal-secret and stores it here. Cleared when
// the user toggles eye off, or on every reload — never persists.
const [revealedSecrets, setRevealedSecrets] = useState<Record<string, string>>({})
const [editMode, setEditMode] = useState(false)
const [hasChanges, setHasChanges] = useState(false)
const [expandedCategories, setExpandedCategories] = useState<Set<string>>(new Set())
@@ -1065,8 +1071,86 @@ export function NotificationSettings() {
}
}
const toggleSecret = (key: string) => {
setShowSecrets(prev => ({ ...prev, [key]: !prev[key] }))
// Maps each eye-button local key to the body shape the backend
// expects for /api/notifications/reveal-secret. Centralised here so
// the JSX call site stays a one-liner — `toggleSecret(key)`.
const SECRET_REVEAL_TARGETS: Record<string, Record<string, string>> = {
tg_token: { channel: "telegram", field: "bot_token" },
gt_token: { channel: "gotify", field: "token" },
dc_hook: { channel: "discord", field: "webhook_url" },
em_pass: { channel: "email", field: "password" },
apprise_url: { channel: "apprise", field: "url" },
}
const toggleSecret = async (key: string) => {
// Turning eye OFF — drop the cached cleartext immediately.
if (showSecrets[key]) {
setShowSecrets(prev => ({ ...prev, [key]: false }))
setRevealedSecrets(prev => {
const next = { ...prev }
delete next[key]
return next
})
return
}
// Turning eye ON — resolve the backend body. ai_key is dynamic
// (depends on current provider); the rest are static.
let body: Record<string, string> | null = null
if (key === "ai_key") {
const prov = (config.ai_provider || "").trim()
if (prov) body = { ai_provider: prov }
} else {
body = SECRET_REVEAL_TARGETS[key] || null
}
if (!body) {
// Unknown eye-toggle — flip state and let the input render
// whatever it already has. Never silently skip the UI feedback.
setShowSecrets(prev => ({ ...prev, [key]: true }))
return
}
try {
// Must go through `fetchApi` (not raw `fetch`) so the JWT
// Authorization header gets attached — the endpoint uses
// `@require_auth` and would reject a plain fetch with 401, which
// is exactly what produced the "still asterisks" behaviour seen
// on the first build.
const data = await fetchApi<{ value?: string }>("/api/notifications/reveal-secret", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
})
const value = typeof data?.value === "string" ? data.value : ""
setRevealedSecrets(prev => ({ ...prev, [key]: value }))
} catch {
// Network/auth failure → still flip the eye so the user sees
// the underlying input changing (and that it has the masked
// placeholder, which signals "not loaded"). Don't crash.
}
setShowSecrets(prev => ({ ...prev, [key]: true }))
}
// Sentinel the backend sends in place of any populated sensitive
// value (see SENSITIVE_PLACEHOLDER in notification_manager.py).
const SECRET_PLACEHOLDER = "************"
// Render value for a secret input:
// - if the eye is on AND the stored value is the masked placeholder
// AND we have a revealed cleartext for this key → show cleartext.
// - otherwise show whatever the input already has (the placeholder
// while masked, or the value the user is typing in editMode).
// This avoids overwriting an in-progress edit when the eye toggles.
const secretValue = (key: string, current: string): string => {
if (
showSecrets[key] &&
current === SECRET_PLACEHOLDER &&
revealedSecrets[key] !== undefined
) {
return revealedSecrets[key]
}
return current
}
if (loading) {
@@ -1402,7 +1486,7 @@ export function NotificationSettings() {
type={showSecrets["tg_token"] ? "text" : "password"}
className={`h-7 text-xs font-mono ${!editMode ? "opacity-50" : ""}`}
placeholder="7595377878:AAGE6Fb2cy... (with or without 'bot' prefix)"
value={config.channels.telegram?.bot_token || ""}
value={secretValue("tg_token", config.channels.telegram?.bot_token || "")}
onChange={e => updateChannel("telegram", "bot_token", e.target.value)}
disabled={!editMode}
/>
@@ -1519,7 +1603,7 @@ export function NotificationSettings() {
type={showSecrets["gt_token"] ? "text" : "password"}
className={`h-7 text-xs font-mono ${!editMode ? "opacity-50" : ""}`}
placeholder="A_valid_gotify_token"
value={config.channels.gotify?.token || ""}
value={secretValue("gt_token", config.channels.gotify?.token || "")}
onChange={e => updateChannel("gotify", "token", e.target.value)}
disabled={!editMode}
/>
@@ -1597,7 +1681,7 @@ export function NotificationSettings() {
type={showSecrets["dc_hook"] ? "text" : "password"}
className={`h-7 text-xs font-mono ${!editMode ? "opacity-50" : ""}`}
placeholder="https://discord.com/api/webhooks/..."
value={config.channels.discord?.webhook_url || ""}
value={secretValue("dc_hook", config.channels.discord?.webhook_url || "")}
onChange={e => updateChannel("discord", "webhook_url", e.target.value)}
disabled={!editMode}
/>
@@ -1729,7 +1813,7 @@ export function NotificationSettings() {
type={showSecrets["em_pass"] ? "text" : "password"}
className={`h-7 text-xs font-mono ${!editMode ? "opacity-50" : ""}`}
placeholder="App password"
value={config.channels.email?.password || ""}
value={secretValue("em_pass", config.channels.email?.password || "")}
onChange={e => updateChannel("email", "password", e.target.value)}
disabled={!editMode}
/>
@@ -1840,14 +1924,14 @@ export function NotificationSettings() {
type={showSecrets["apprise_url"] ? "text" : "password"}
className={`h-7 text-xs font-mono min-w-0 flex-1 ${!editMode ? "opacity-50" : ""}`}
placeholder="tgram://bottoken/ChatID"
value={config.channels.apprise?.url || ""}
value={secretValue("apprise_url", config.channels.apprise?.url || "")}
onChange={e => updateChannel("apprise", "url", e.target.value)}
disabled={!editMode}
/>
<button
type="button"
className="h-7 w-7 shrink-0 flex items-center justify-center rounded-md border border-border hover:bg-muted text-muted-foreground"
onClick={() => setShowSecrets(s => ({ ...s, apprise_url: !s.apprise_url }))}
onClick={() => toggleSecret("apprise_url")}
title={showSecrets["apprise_url"] ? "Hide URL" : "Show URL"}
>
{showSecrets["apprise_url"] ? <EyeOff className="h-3 w-3" /> : <Eye className="h-3 w-3" />}
@@ -2122,13 +2206,13 @@ export function NotificationSettings() {
type={showSecrets["ai_key"] ? "text" : "password"}
className="h-9 text-sm font-mono"
placeholder="sk-..."
value={config.ai_api_keys?.[config.ai_provider] || ""}
onChange={e => updateConfig(p => ({
...p,
ai_api_keys: {
...p.ai_api_keys,
[p.ai_provider]: e.target.value
}
value={secretValue("ai_key", config.ai_api_keys?.[config.ai_provider] || "")}
onChange={e => updateConfig(p => ({
...p,
ai_api_keys: {
...p.ai_api_keys,
[p.ai_provider]: e.target.value
}
}))}
disabled={!editMode}
/>
@@ -0,0 +1,279 @@
"use client"
import { useEffect, useState } from "react"
import { Dialog, DialogContent, DialogHeader, DialogTitle, DialogDescription } from "./ui/dialog"
import { Input } from "./ui/input"
import { ScrollArea } from "./ui/scroll-area"
import { Cpu, MemoryStick, Search } from "lucide-react"
import { fetchApi } from "@/lib/api-config"
import { ProcessInfoModal } from "./process-info-modal"
interface ProcessInfo {
pid: number
/** Parent process PID — equal to `pid` for process rows, different
* for thread rows (CPU sort enumerates per-thread). The detail modal
* always loads the parent, since /proc/<pid>/cmdline etc. only exist
* at the process level. */
parent_pid?: number
user: string
cpu: number
/** CPU% averaged over the process's whole lifetime. Surfaces
* long-running idle processes that consume a steady baseline but
* don't spike during the 1-s sample window (an orphan `bash -s`
* in a sleep-loop, a polling daemon, etc.). Only meaningful on
* the CPU sort response. */
cpu_avg?: number
mem: number
rss_kb: number
command: string
/** Full command line. Used for filter matching and hover tooltips so
* searching e.g. "proxmenux" finds a process whose short name is just
* "python3" but whose cmdline is `python3 /.../proxmenux.py`. */
cmdline?: string
}
interface ProcessesResponse {
processes: ProcessInfo[]
sort: "cpu" | "mem"
captured_at: number
}
interface ProcessDetailModalProps {
open: boolean
onOpenChange: (open: boolean) => void
/** Which metric the parent card represents (drives default sort + emphasis) */
sort: "cpu" | "mem"
}
const REFRESH_MS = 3000
// FETCH_LIMIT is how many rows the server returns. DISPLAY_LIMIT is what
// the user actually sees when no filter is set. We over-fetch so the
// filter can find processes that aren't in the top-N by metric — e.g.,
// searching "proxmenux" in the Memory modal should find it even though
// it's nowhere near the top 25 by RSS.
const FETCH_LIMIT = 200
const DISPLAY_LIMIT = 25
const formatRss = (kb: number): string => {
if (kb >= 1024 * 1024) return `${(kb / 1024 / 1024).toFixed(2)} GB`
if (kb >= 1024) return `${(kb / 1024).toFixed(1)} MB`
return `${kb} KB`
}
export function ProcessDetailModal({ open, onOpenChange, sort }: ProcessDetailModalProps) {
const [data, setData] = useState<ProcessesResponse | null>(null)
const [error, setError] = useState<string | null>(null)
const [loading, setLoading] = useState(false)
const [filter, setFilter] = useState("")
const [selectedPid, setSelectedPid] = useState<number | null>(null)
const fetchProcesses = async (silent = false) => {
if (!silent) setLoading(true)
setError(null)
try {
const res = await fetchApi<ProcessesResponse>(`/api/processes?sort=${sort}&limit=${FETCH_LIMIT}`)
setData(res)
} catch (e: any) {
setError(e?.message || "Failed to fetch processes")
} finally {
if (!silent) setLoading(false)
}
}
useEffect(() => {
if (!open) return
fetchProcesses()
const id = setInterval(() => fetchProcesses(true), REFRESH_MS)
return () => clearInterval(id)
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [open, sort])
// Reset filter when dialog closes
useEffect(() => {
if (!open) setFilter("")
}, [open])
// When no filter is set, the user just wants the top N by metric
// (CPU usage or memory). When they type a query, they want EVERY
// match — including processes that aren't in the top N — which is
// why we over-fetch on the server.
const allMatches = (data?.processes ?? []).filter((p) => {
if (!filter) return true
const q = filter.toLowerCase()
return (
p.command.toLowerCase().includes(q) ||
(p.cmdline?.toLowerCase().includes(q) ?? false) ||
p.user.toLowerCase().includes(q) ||
String(p.pid).includes(q)
)
})
const filtered = filter ? allMatches : allMatches.slice(0, DISPLAY_LIMIT)
const Icon = sort === "cpu" ? Cpu : MemoryStick
const title = sort === "cpu" ? "Top processes by CPU" : "Top processes by Memory"
const description =
sort === "cpu"
? "Current CPU usage per process, as a fraction of the host's total CPU — same scale as the CPU Usage card above. Refreshes every 3 s while open."
: "Current resident memory per process. Refreshes every 3 s while open."
// Accent palette matched to the Overview cards: CPU Usage donut uses
// blue (#3b82f6), Memory cached uses rgba(99,102,241,0.55) — we keep
// the same hues so the modal feels like a continuation of the card.
const accent = sort === "cpu"
? { dot: "#3b82f6", bar: "#3b82f6", text: "text-blue-500" }
: { dot: "#6366f1", bar: "#6366f1", text: "text-indigo-400" }
// Scale bars to the largest value in the (filtered) list so the visual
// ranking is preserved even when no process is near 100 %. CPU can
// exceed 100 % on multi-threaded apps — falling back to max=1 prevents
// a divide-by-zero when the list is empty.
const maxPrimary = Math.max(
1,
...filtered.map((p) => (sort === "cpu" ? p.cpu : p.mem))
)
// Mobile drops PID + USER; desktop keeps the full 5-column layout.
// CPU and MEM columns are wider on desktop with a real gap between
// them so the two metrics don't feel glued together.
const gridCols =
"grid-cols-[minmax(0,1fr)_70px_90px] sm:grid-cols-[60px_96px_minmax(140px,1fr)_110px_120px]"
return (
<>
<Dialog open={open} onOpenChange={onOpenChange}>
<DialogContent
className="max-w-3xl"
/* Prevent Radix from focusing the search Input on open — the
auto-focus pops the on-screen keyboard on touch devices and
covers half the modal. The user can still tap the field to
start filtering. */
onOpenAutoFocus={(e) => e.preventDefault()}
>
<DialogHeader>
<DialogTitle className="flex items-center gap-2">
<Icon className={`h-5 w-5 ${accent.text}`} />
{title}
</DialogTitle>
<DialogDescription className="text-xs">{description}</DialogDescription>
</DialogHeader>
<div className="relative mb-2">
<Search className="absolute left-2 top-1/2 -translate-y-1/2 h-4 w-4 text-muted-foreground" />
<Input
placeholder="Filter by command line, user or PID..."
value={filter}
onChange={(e) => setFilter(e.target.value)}
className="pl-8 h-8 text-sm"
/>
</div>
{error ? (
<div className="text-sm text-red-500 py-4">{error}</div>
) : (
<ScrollArea className="h-[440px] border border-border rounded-md">
<div className="min-w-full">
{/* Sticky solid header so scrolled rows don't bleed through */}
<div
className={`grid items-center gap-x-3 sm:gap-x-6 px-3 py-2 text-[10px] font-medium uppercase tracking-wider text-muted-foreground border-b border-border bg-card sticky top-0 z-10 ${gridCols}`}
>
<div className="hidden sm:block">PID</div>
<div className="hidden sm:block truncate">User</div>
<div>Command</div>
<div className={`text-right ${sort === "cpu" ? accent.text : ""}`}>CPU %</div>
<div className={`text-right ${sort === "mem" ? accent.text : ""}`}>{sort === "mem" ? "Memory" : "Mem %"}</div>
</div>
{filtered.length === 0 && !loading ? (
<div className="text-center py-8 text-sm text-muted-foreground">
No processes match the filter
</div>
) : (
filtered.map((p) => {
const primary = sort === "cpu" ? p.cpu : p.mem
const barPct = Math.min(100, (primary / maxPrimary) * 100)
return (
<button
key={p.pid}
type="button"
onClick={() => setSelectedPid(p.parent_pid ?? p.pid)}
className={`w-full text-left grid items-center gap-x-3 sm:gap-x-6 px-3 py-2 border-b border-border/40 hover:bg-white/5 transition-colors ${gridCols}`}
>
<div className="hidden sm:flex font-mono text-xs items-center gap-1.5 min-w-0">
<span
className="w-1.5 h-1.5 rounded-full flex-shrink-0"
style={{ background: accent.dot }}
/>
<span className="truncate">{p.pid}</span>
</div>
<div className="hidden sm:block font-mono text-xs truncate" title={p.user}>{p.user}</div>
<div className="font-mono text-xs truncate min-w-0 flex items-center gap-1.5" title={p.cmdline || p.command}>
{/* Mobile only: keep the accent dot since PID column is gone */}
<span
className="sm:hidden w-1.5 h-1.5 rounded-full flex-shrink-0"
style={{ background: accent.dot }}
/>
<span className="truncate">{p.command}</span>
</div>
{/* Primary metric: value + sized progress bar in the accent colour */}
{sort === "cpu" ? (
<div className="flex flex-col items-end gap-1 min-w-0">
<span className={`font-mono text-sm font-semibold ${accent.text}`}>{p.cpu.toFixed(1)}</span>
{/* Show lifetime average only when it
materially differs from the live
sample (process consumes a steady
baseline but was idle at sample time
— orphaned bash loop, polling daemon).
The 0.5 / 1.5x thresholds skip cases
where avg and now match within sampler
noise. */}
{typeof p.cpu_avg === "number" && p.cpu_avg >= 0.5 && p.cpu_avg > p.cpu * 1.5 && (
<span className="font-mono text-[10px] text-amber-400" title="Average CPU% across this process's lifetime — useful for finding long-running idle baselines">
avg {p.cpu_avg.toFixed(1)}
</span>
)}
<div className="w-full h-1 bg-muted rounded-full overflow-hidden">
<div className="h-full rounded-full" style={{ width: `${barPct}%`, background: accent.bar }} />
</div>
</div>
) : (
<div className="font-mono text-xs text-right text-muted-foreground">{p.cpu.toFixed(1)}</div>
)}
{/* Secondary column: mem % when CPU is primary, RSS when memory is primary */}
{sort === "cpu" ? (
<div className="font-mono text-xs text-right text-muted-foreground">{p.mem.toFixed(1)}</div>
) : (
<div className="flex flex-col items-end gap-1 min-w-0">
<span className={`font-mono text-sm font-semibold ${accent.text}`}>{formatRss(p.rss_kb)}</span>
<div className="w-full h-1 bg-muted rounded-full overflow-hidden">
<div className="h-full rounded-full" style={{ width: `${barPct}%`, background: accent.bar }} />
</div>
</div>
)}
</button>
)
})
)}
</div>
</ScrollArea>
)}
{data?.captured_at && (
<div className="text-[10px] text-muted-foreground text-right mt-1">
Captured {new Date(data.captured_at * 1000).toLocaleTimeString()} · {filter
? `${allMatches.length} match${allMatches.length === 1 ? '' : 'es'} of ${data.processes.length} processes`
: `Top ${filtered.length} of ${data.processes.length} processes`}
</div>
)}
</DialogContent>
</Dialog>
<ProcessInfoModal
pid={selectedPid}
accent={accent}
onClose={() => setSelectedPid(null)}
/>
</>
)
}
+259
View File
@@ -0,0 +1,259 @@
"use client"
import { useEffect, useRef, useState } from "react"
import { Dialog, DialogContent, DialogHeader, DialogTitle, DialogDescription } from "./ui/dialog"
import { ScrollArea } from "./ui/scroll-area"
import { Activity, FileText, HardDrive, Clock, Info } from "lucide-react"
import { fetchApi } from "@/lib/api-config"
interface ProcessDetail {
pid: number
comm: string
cmdline: string
exe: string | null
cwd: string | null
state: string
ppid: number
parent_name: string | null
threads: number
vm_rss_kb: number
vm_size_kb: number
vm_swap_kb: number
user: string
group: string
uid: number
gid: number
start_time: string | null
elapsed: string | null
cpu: number
mem: number
io_read_bytes: number | null
io_write_bytes: number | null
fd_count: number | null
captured_at: number
}
interface ProcessInfoModalProps {
pid: number | null
accent: { dot: string; bar: string; text: string }
onClose: () => void
}
const REFRESH_MS = 3000
const formatKb = (kb: number | null | undefined): string => {
if (kb == null) return "—"
if (kb >= 1024 * 1024) return `${(kb / 1024 / 1024).toFixed(2)} GB`
if (kb >= 1024) return `${(kb / 1024).toFixed(1)} MB`
return `${kb} KB`
}
const formatBytes = (b: number | null | undefined): string => {
if (b == null) return "—"
if (b >= 1024 * 1024 * 1024) return `${(b / 1024 / 1024 / 1024).toFixed(2)} GB`
if (b >= 1024 * 1024) return `${(b / 1024 / 1024).toFixed(1)} MB`
if (b >= 1024) return `${(b / 1024).toFixed(1)} KB`
return `${b} B`
}
// Linux process states from /proc/<pid>/status. The first char of `State:`
// is the canonical letter — the rest of the field is a human label like
// "(running)". We expand the bare letter to something readable.
const stateLabel = (state: string): string => {
const letter = (state || "").trim().charAt(0).toUpperCase()
const map: Record<string, string> = {
R: "Running",
S: "Sleeping",
D: "Disk wait",
Z: "Zombie",
T: "Stopped",
t: "Tracing stop",
X: "Dead",
I: "Idle",
}
return map[letter] || state || "—"
}
export function ProcessInfoModal({ pid, accent, onClose }: ProcessInfoModalProps) {
const [data, setData] = useState<ProcessDetail | null>(null)
const [error, setError] = useState<string | null>(null)
const [loading, setLoading] = useState(false)
const [exited, setExited] = useState(false)
const intervalRef = useRef<ReturnType<typeof setInterval> | null>(null)
const open = pid != null
const stopPolling = () => {
if (intervalRef.current) {
clearInterval(intervalRef.current)
intervalRef.current = null
}
}
const fetchDetail = async (silent = false) => {
if (pid == null) return
if (!silent) setLoading(true)
setError(null)
try {
const res = await fetchApi<ProcessDetail>(`/api/processes/${pid}`)
setData(res)
} catch (e: any) {
// 404 = the process exited while the modal was open. Expected for
// short-lived helpers (pct exec, backup subprocesses, the `ps` snapshot
// itself). Keep the last good snapshot on screen, stop polling, and
// surface an info banner — NOT an error — so it doesn't look like a bug.
if (e?.message?.includes("404")) {
setExited(true)
stopPolling()
} else {
setError(e?.message || "Failed to fetch process")
}
} finally {
if (!silent) setLoading(false)
}
}
useEffect(() => {
if (pid == null) {
setData(null)
setError(null)
setExited(false)
stopPolling()
return
}
setExited(false)
fetchDetail()
intervalRef.current = setInterval(() => fetchDetail(true), REFRESH_MS)
return () => stopPolling()
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [pid])
return (
<Dialog open={open} onOpenChange={(v) => { if (!v) onClose() }}>
<DialogContent className="max-w-2xl">
<DialogHeader>
<DialogTitle className="flex items-center gap-2 min-w-0">
<span
className="w-2 h-2 rounded-full flex-shrink-0"
style={{ background: accent.dot }}
/>
<span className="truncate font-mono text-base">{data?.comm || "Process"}</span>
<span className="text-xs text-muted-foreground font-mono flex-shrink-0">PID {pid}</span>
</DialogTitle>
<DialogDescription className="text-xs">
{exited ? (
<>Last snapshot from <span className="font-mono">/proc/{pid}</span> before the process finished.</>
) : (
<>Live snapshot from <span className="font-mono">/proc/{pid}</span>. Auto-refreshes every {REFRESH_MS / 1000} s while open.</>
)}
</DialogDescription>
</DialogHeader>
{/* Info banner when the process has finished. Amber, not red — this is
expected behavior for short-lived processes, not an error. */}
{exited && (
<div className="flex items-start gap-2 px-3 py-2 rounded-md border border-amber-500/30 bg-amber-500/10 text-xs text-amber-300">
<Info className="h-4 w-4 flex-shrink-0 mt-0.5" />
<div>
<div className="font-medium text-amber-200">This process has finished</div>
<div className="text-amber-300/80 mt-0.5">
It was likely a short-lived helper (a script, a <span className="font-mono">pct exec</span>, or a one-shot command) that completed while the modal was open. The data below is the last snapshot captured before it exited not a stale or broken read.
</div>
</div>
</div>
)}
{error && !data ? (
<div className="text-sm text-red-500 py-4">{error}</div>
) : !data ? (
<div className="text-sm text-muted-foreground py-8 text-center">
{loading ? "Loading…" : "—"}
</div>
) : (
<ScrollArea className={`max-h-[480px] pr-2 ${exited ? "opacity-75" : ""}`}>
<div className="space-y-4">
{/* Overview */}
<Section icon={<Activity className="h-4 w-4 text-blue-400" />} title="Overview">
<Row label="State" value={exited ? "Exited" : stateLabel(data.state)} />
<Row label="Parent" value={data.parent_name ? `${data.parent_name} (PID ${data.ppid})` : `PID ${data.ppid}`} mono />
<Row label="Threads" value={String(data.threads)} mono />
<Row label="Open FDs" value={data.fd_count != null ? String(data.fd_count) : "—"} mono />
<Row label="User" value={`${data.user} (${data.uid})`} mono />
<Row label="Group" value={`${data.group} (${data.gid})`} mono />
</Section>
{/* Resources */}
<Section icon={<HardDrive className="h-4 w-4 text-amber-400" />} title="Resources">
<Row label="CPU" value={`${data.cpu.toFixed(1)} %`} mono valueClass={accent.text} />
<Row label="Memory" value={`${data.mem.toFixed(1)} %`} mono valueClass={accent.text} />
<Row label="Resident (RSS)" value={formatKb(data.vm_rss_kb)} mono />
<Row label="Virtual size" value={formatKb(data.vm_size_kb)} mono />
<Row label="Swap" value={formatKb(data.vm_swap_kb)} mono />
<Row label="I/O read" value={formatBytes(data.io_read_bytes)} mono />
<Row label="I/O write" value={formatBytes(data.io_write_bytes)} mono />
</Section>
{/* Command */}
<Section icon={<FileText className="h-4 w-4 text-purple-400" />} title="Command">
<Row label="Name" value={data.comm} mono />
<Row label="Command line" value={data.cmdline || data.comm} mono wrap />
<Row label="Executable" value={data.exe || "—"} mono wrap />
<Row label="Working dir" value={data.cwd || "—"} mono wrap />
</Section>
{/* Times */}
<Section icon={<Clock className="h-4 w-4 text-emerald-400" />} title="Lifetime">
<Row label="Started" value={data.start_time || "—"} mono />
<Row label="Running for" value={data.elapsed || "—"} mono />
</Section>
</div>
</ScrollArea>
)}
{data?.captured_at && (
<div className="text-[10px] text-muted-foreground text-right mt-1">
{exited ? "Last seen" : "Captured"} {new Date(data.captured_at * 1000).toLocaleTimeString()}
{error ? ` · ${error}` : ""}
</div>
)}
</DialogContent>
</Dialog>
)
}
function Section({ icon, title, children }: { icon: React.ReactNode; title: string; children: React.ReactNode }) {
return (
<div className="border border-border rounded-md overflow-hidden">
<div className="flex items-center gap-2 px-3 py-2 bg-card text-xs font-medium uppercase tracking-wider text-muted-foreground border-b border-border">
{icon}
{title}
</div>
<div className="divide-y divide-border/40">{children}</div>
</div>
)
}
function Row({
label,
value,
mono,
wrap,
valueClass,
}: {
label: string
value: string
mono?: boolean
wrap?: boolean
valueClass?: string
}) {
return (
<div className="grid grid-cols-[110px_minmax(0,1fr)] gap-2 px-3 py-1.5 text-xs">
<div className="text-muted-foreground">{label}</div>
<div
className={`${mono ? "font-mono" : ""} ${wrap ? "break-all" : "truncate"} ${valueClass || ""}`}
title={value}
>
{value}
</div>
</div>
)
}
+218 -240
View File
@@ -14,6 +14,7 @@ import { Settings } from "./settings"
import { Security } from "./security"
import { Profile } from "./profile"
import { About } from "./about"
import { HostBackup } from "./host-backup"
import { OnboardingCarousel } from "./onboarding-carousel"
import { HealthStatusModal } from "./health-status-modal"
import { ReleaseNotesModal, useVersionCheck } from "./release-notes-modal"
@@ -30,17 +31,26 @@ import {
LayoutDashboard,
HardDrive,
NetworkIcon,
Box,
Boxes,
Cpu,
FileText,
ScrollText,
SettingsIcon,
Settings2,
Terminal,
ShieldCheck,
Info,
DatabaseBackup,
ChevronDown,
} from "lucide-react"
import Image from "next/image"
import { ThemeToggle } from "./theme-toggle"
import { Sheet, SheetContent, SheetTrigger } from "./ui/sheet"
import {
DropdownMenu,
DropdownMenuContent,
DropdownMenuItem,
DropdownMenuTrigger,
} from "./ui/dropdown-menu"
interface SystemStatus {
status: "healthy" | "warning" | "critical"
@@ -352,28 +362,19 @@ export function ProxmoxDashboard() {
const getActiveTabLabel = () => {
switch (activeTab) {
case "overview":
return "Overview"
case "storage":
return "Storage"
case "network":
return "Network"
case "vms":
return "VMs & LXCs"
case "hardware":
return "Hardware"
case "terminal":
return "Terminal"
case "logs":
return "System Logs"
case "security":
return "Security"
case "settings":
return "Settings"
case "profile":
return "Profile"
default:
return "Navigation Menu"
case "overview": return "Overview"
case "vms": return "VMs & LXCs"
case "storage": return "Storage"
case "network": return "Network"
case "hardware": return "Hardware"
case "backup": return "Backup"
case "terminal": return "Terminal"
case "logs": return "System Logs"
case "security": return "Security"
case "settings": return "Settings"
case "about": return "About"
case "profile": return "Profile"
default: return "Navigation Menu"
}
}
@@ -565,71 +566,128 @@ export function ProxmoxDashboard() {
>
<div className="container mx-auto px-4 lg:px-6 pt-4 lg:pt-6">
<Tabs value={activeTab} onValueChange={setActiveTab} className="space-y-0">
{/* Issue #191: 10 tabs after adding About. The grid wraps via
Tabs primitives so the extra column doesn't push the
triggers off-screen on common laptop widths. */}
<TabsList className="hidden lg:grid w-full grid-cols-10 bg-card border border-border">
<TabsTrigger
value="overview"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
Overview
</TabsTrigger>
<TabsTrigger
value="storage"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
Storage
</TabsTrigger>
<TabsTrigger
value="network"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
Network
</TabsTrigger>
<TabsTrigger
value="vms"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
VMs & LXCs
</TabsTrigger>
<TabsTrigger
value="hardware"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
Hardware
</TabsTrigger>
<TabsTrigger
value="logs"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
System Logs
</TabsTrigger>
<TabsTrigger
value="terminal"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
Terminal
</TabsTrigger>
<TabsTrigger
value="security"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
Security
</TabsTrigger>
<TabsTrigger
value="settings"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
Settings
</TabsTrigger>
<TabsTrigger
value="about"
className="data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
>
About
</TabsTrigger>
</TabsList>
{/* Sprint 13D nav redesign — 6 top-level slots in usage order:
Overview · VMs & LXCs · Node ⌄ · Backup · Terminal · Admin ⌄
Node groups Storage / Network / Hardware (3 sub-items).
Admin groups System Logs / Security / Settings / About
(will split when RBAC arrives in 1.5.0).
Backup is direct now (only Host Backup); becomes a dropdown
when VM/LXC centralised backup ships. */}
{(() => {
const triggerActiveClass =
"data-[state=active]:bg-blue-500 data-[state=active]:text-white data-[state=active]:rounded-md"
// Each dropdown lists its children in the order they
// render. When one of them is the active tab, the dropdown
// trigger swaps its label + icon to that child — same
// pattern macOS Settings uses inside a category: the
// crumb shows where you are, the chevron tells you the
// siblings are one click away.
const NODE_ITEMS = [
{ value: "storage", label: "Storage", Icon: HardDrive, default: false },
{ value: "network", label: "Network", Icon: NetworkIcon, default: false },
{ value: "hardware", label: "Hardware", Icon: Cpu, default: false },
]
const ADMIN_ITEMS = [
{ value: "logs", label: "System Logs", Icon: ScrollText, default: false },
{ value: "security", label: "Security", Icon: ShieldCheck, default: false },
{ value: "settings", label: "Settings", Icon: SettingsIcon, default: false },
{ value: "about", label: "About", Icon: Info, default: false },
]
const activeNodeItem = NODE_ITEMS.find(i => i.value === activeTab)
const activeAdminItem = ADMIN_ITEMS.find(i => i.value === activeTab)
const isNodeActive = activeNodeItem !== undefined
const isAdminActive = activeAdminItem !== undefined
// The trigger label + icon shown on the bar. When a child
// is active we surface IT; otherwise the group default.
const NodeTriggerIcon = activeNodeItem ? activeNodeItem.Icon : Server
const NodeTriggerLabel = activeNodeItem ? activeNodeItem.label : "Node"
const AdminTriggerIcon = activeAdminItem ? activeAdminItem.Icon : Settings2
const AdminTriggerLabel = activeAdminItem ? activeAdminItem.label : "Admin"
// Dropdown trigger styling: parity with TabsTrigger so the
// parent visibly carries the "I'm the selected section"
// signal when any of its children is the active tab —
// same blue background + white text + rounded as a direct
// tab. Without this the user lands on Storage and the
// entire top bar looks idle.
const dropdownBtnClass = (active: boolean) =>
`inline-flex items-center justify-center whitespace-nowrap px-3 py-1.5 text-sm font-medium ring-offset-background transition-all focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 disabled:pointer-events-none disabled:opacity-50 ${
active
? "bg-blue-500 text-white rounded-md"
: "text-muted-foreground hover:text-foreground rounded-sm"
}`
return (
<TabsList className="hidden lg:grid w-full grid-cols-6 bg-card border border-border">
{/* Direct: Overview */}
<TabsTrigger value="overview" className={triggerActiveClass}>
<LayoutDashboard className="mr-2 h-4 w-4" />
Overview
</TabsTrigger>
{/* Direct: VMs & LXCs — first-class because Proxmox IS
a hypervisor; workloads belong at top level. */}
<TabsTrigger value="vms" className={triggerActiveClass}>
<Boxes className="mr-2 h-4 w-4" />
VMs &amp; LXCs
</TabsTrigger>
{/* Dropdown: Node (Storage / Network / Hardware) */}
<DropdownMenu>
<DropdownMenuTrigger className={dropdownBtnClass(isNodeActive)}>
<NodeTriggerIcon className="mr-2 h-4 w-4" />
{NodeTriggerLabel}
<ChevronDown className="ml-1.5 h-3 w-3 opacity-70" />
</DropdownMenuTrigger>
<DropdownMenuContent align="center" className="min-w-[180px]">
{NODE_ITEMS.map(({ value, label, Icon }) => (
<DropdownMenuItem
key={value}
onClick={() => setActiveTab(value)}
className={activeTab === value ? "bg-blue-500/10 text-blue-500" : ""}
>
<Icon className="mr-2 h-4 w-4" />
{label}
</DropdownMenuItem>
))}
</DropdownMenuContent>
</DropdownMenu>
{/* Direct: Backup (today: Host Backup only). When VM/LXC
backup ships this becomes a dropdown. */}
<TabsTrigger value="backup" className={triggerActiveClass}>
<DatabaseBackup className="mr-2 h-4 w-4" />
Backup
</TabsTrigger>
{/* Direct: Terminal */}
<TabsTrigger value="terminal" className={triggerActiveClass}>
<Terminal className="mr-2 h-4 w-4" />
Terminal
</TabsTrigger>
{/* Dropdown: Admin (System Logs / Security / Settings / About) */}
<DropdownMenu>
<DropdownMenuTrigger className={dropdownBtnClass(isAdminActive)}>
<AdminTriggerIcon className="mr-2 h-4 w-4" />
{AdminTriggerLabel}
<ChevronDown className="ml-1.5 h-3 w-3 opacity-70" />
</DropdownMenuTrigger>
<DropdownMenuContent align="center" className="min-w-[180px]">
{ADMIN_ITEMS.map(({ value, label, Icon }) => (
<DropdownMenuItem
key={value}
onClick={() => setActiveTab(value)}
className={activeTab === value ? "bg-blue-500/10 text-blue-500" : ""}
>
<Icon className="mr-2 h-4 w-4" />
{label}
</DropdownMenuItem>
))}
</DropdownMenuContent>
</DropdownMenu>
</TabsList>
)
})()}
<Sheet open={mobileMenuOpen} onOpenChange={setMobileMenuOpen}>
<div className="lg:hidden">
@@ -646,158 +704,74 @@ export function ProxmoxDashboard() {
</SheetTrigger>
</div>
<SheetContent side="top" className="bg-card border-border">
<div className="flex flex-col gap-2 mt-4">
<Button
variant="ghost"
onClick={() => {
setActiveTab("overview")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "overview"
{(() => {
// Sheet items mirror the desktop layout: 6 sections,
// with two of them (Node, Admin) collapsing into a
// header + nested items. Direct tabs (Overview, VMs,
// Backup, Terminal) sit at the top level.
const select = (v: string) => {
setActiveTab(v)
setMobileMenuOpen(false)
}
const itemClass = (active: boolean) =>
`w-full justify-start gap-3 ${
active
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<LayoutDashboard className="h-5 w-5" />
<span>Overview</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("storage")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "storage"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<HardDrive className="h-5 w-5" />
<span>Storage</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("network")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "network"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<NetworkIcon className="h-5 w-5" />
<span>Network</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("vms")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "vms"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<Box className="h-5 w-5" />
<span>VMs & LXCs</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("hardware")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "hardware"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<Cpu className="h-5 w-5" />
<span>Hardware</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("logs")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "logs"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<FileText className="h-5 w-5" />
<span>System Logs</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("terminal")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "terminal"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<Terminal className="h-5 w-5" />
<span>Terminal</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("security")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "security"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<ShieldCheck className="h-5 w-5" />
<span>Security</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("settings")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "settings"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<SettingsIcon className="h-5 w-5" />
<span>Settings</span>
</Button>
<Button
variant="ghost"
onClick={() => {
setActiveTab("about")
setMobileMenuOpen(false)
}}
className={`w-full justify-start gap-3 ${
activeTab === "about"
? "bg-blue-500/10 text-blue-500 border-l-4 border-blue-500 rounded-l-none"
: ""
}`}
>
<Info className="h-5 w-5" />
<span>About</span>
</Button>
</div>
}`
// Mobile sheet is a flat list (no section headers).
// The desktop layout uses dropdowns to express the
// Node/Admin grouping; here we just enumerate items
// in the same visual order.
return (
<div className="flex flex-col gap-1 mt-4">
<Button variant="ghost" onClick={() => select("overview")} className={itemClass(activeTab === "overview")}>
<LayoutDashboard className="h-5 w-5" />
<span>Overview</span>
</Button>
<Button variant="ghost" onClick={() => select("vms")} className={itemClass(activeTab === "vms")}>
<Boxes className="h-5 w-5" />
<span>VMs &amp; LXCs</span>
</Button>
<Button variant="ghost" onClick={() => select("storage")} className={itemClass(activeTab === "storage")}>
<HardDrive className="h-5 w-5" />
<span>Storage</span>
</Button>
<Button variant="ghost" onClick={() => select("network")} className={itemClass(activeTab === "network")}>
<NetworkIcon className="h-5 w-5" />
<span>Network</span>
</Button>
<Button variant="ghost" onClick={() => select("hardware")} className={itemClass(activeTab === "hardware")}>
<Cpu className="h-5 w-5" />
<span>Hardware</span>
</Button>
<Button variant="ghost" onClick={() => select("backup")} className={itemClass(activeTab === "backup")}>
<DatabaseBackup className="h-5 w-5" />
<span>Backup</span>
</Button>
<Button variant="ghost" onClick={() => select("terminal")} className={itemClass(activeTab === "terminal")}>
<Terminal className="h-5 w-5" />
<span>Terminal</span>
</Button>
<Button variant="ghost" onClick={() => select("logs")} className={itemClass(activeTab === "logs")}>
<ScrollText className="h-5 w-5" />
<span>System Logs</span>
</Button>
<Button variant="ghost" onClick={() => select("security")} className={itemClass(activeTab === "security")}>
<ShieldCheck className="h-5 w-5" />
<span>Security</span>
</Button>
<Button variant="ghost" onClick={() => select("settings")} className={itemClass(activeTab === "settings")}>
<SettingsIcon className="h-5 w-5" />
<span>Settings</span>
</Button>
<Button variant="ghost" onClick={() => select("about")} className={itemClass(activeTab === "about")}>
<Info className="h-5 w-5" />
<span>About</span>
</Button>
</div>
)
})()}
</SheetContent>
</Sheet>
</Tabs>
@@ -830,6 +804,10 @@ export function ProxmoxDashboard() {
<SystemLogs key={`logs-${componentKey}`} />
</TabsContent>
<TabsContent value="backup" className="space-y-4 md:space-y-6 mt-0">
<HostBackup key={`backup-${componentKey}`} />
</TabsContent>
<TabsContent value="terminal" className="mt-0">
<TerminalPanel key={`terminal-${componentKey}`} />
</TabsContent>
@@ -858,7 +836,7 @@ export function ProxmoxDashboard() {
</Tabs>
<footer className="mt-8 md:mt-12 pt-4 md:pt-6 border-t border-border text-center text-xs md:text-sm text-muted-foreground">
<p className="font-medium mb-2">ProxMenux Monitor v1.2.2.1-beta</p>
<p className="font-medium mb-2">ProxMenux Monitor v1.2.4</p>
<p>
<a
href="https://ko-fi.com/macrimi"
+220
View File
@@ -0,0 +1,220 @@
"use client"
import { useCallback, useEffect, useState } from "react"
import { Plus, Share, X } from "lucide-react"
// ==========================================================
// PwaInstallPrompt
// ==========================================================
// Bottom-sheet shown on mobile when the Monitor is opened in
// a browser (not launched as an installed PWA). Two variants:
// iOS Safari → manual 3-step instructions
// Android → generic "browser menu → Add to Home Screen"
//
// No `beforeinstallprompt` handling: capturing the event to
// drive a custom Install button interacts badly with Chrome's
// own "Add to Home Screen" menu — Chromium degrades the manual
// path to a plain shortcut when a page has intercepted the
// event but hasn't yet called `prompt()`. Installation goes
// through the browser's own menu entry and produces a real PWA.
//
// Never shown on desktop, or when already running standalone.
// Dismissal options:
// "Not now" → temporary, hidden for 30 days
// "Don't show again" → permanent (no expiry)
// Backdrop / X → session-only dismiss (reappears on
// the next page load)
// ==========================================================
const DISMISSED_FOREVER_KEY = "proxmenux-install-dismissed"
const DISMISSED_UNTIL_KEY = "proxmenux-install-dismissed-until"
const NOT_NOW_DAYS = 30
function isMobileDevice(): boolean {
if (typeof window === "undefined") return false
// Prefer feature detection (coarse pointer + touch) over UA sniffing,
// and fall back to UA for the corner case where a mobile browser
// reports fine pointer under a desktop-mode toggle.
const coarse = window.matchMedia("(pointer: coarse)").matches
const ua = navigator.userAgent
const uaMobile = /Android|iPhone|iPad|iPod|Mobile|Opera Mini|BlackBerry|IEMobile/i.test(ua)
return coarse || uaMobile
}
function isStandalone(): boolean {
if (typeof window === "undefined") return false
const displayModeStandalone = window.matchMedia("(display-mode: standalone)").matches
const iosStandalone = (window.navigator as Navigator & { standalone?: boolean }).standalone === true
return displayModeStandalone || iosStandalone
}
function isIOS(): boolean {
if (typeof window === "undefined") return false
const ua = navigator.userAgent
// iPadOS 13+ reports as MacIntel — detect that too when maxTouchPoints > 1.
const iPadMasqueradingAsMac =
ua.includes("Macintosh") && (navigator as Navigator & { maxTouchPoints?: number }).maxTouchPoints! > 1
return /iPhone|iPad|iPod/i.test(ua) || iPadMasqueradingAsMac
}
export function PwaInstallPrompt() {
const [open, setOpen] = useState(false)
const [platform, setPlatform] = useState<"ios" | "android" | null>(null)
useEffect(() => {
if (typeof window === "undefined") return
if (!isMobileDevice() || isStandalone()) return
try {
if (localStorage.getItem(DISMISSED_FOREVER_KEY) === "1") return
const untilRaw = localStorage.getItem(DISMISSED_UNTIL_KEY)
if (untilRaw) {
const until = Number.parseInt(untilRaw, 10)
// Corrupt / non-numeric values fall through and the prompt shows,
// which is the safe default.
if (Number.isFinite(until) && until > Date.now()) return
}
} catch {
// localStorage unavailable (private mode etc.) — treat as not dismissed.
}
setPlatform(isIOS() ? "ios" : "android")
setOpen(true)
}, [])
const handleNotNow = useCallback(() => {
try {
const until = Date.now() + NOT_NOW_DAYS * 24 * 60 * 60 * 1000
localStorage.setItem(DISMISSED_UNTIL_KEY, String(until))
} catch {
// Best-effort; if localStorage fails the user will see the prompt
// again next visit, which is the safe default.
}
setOpen(false)
}, [])
const handleNeverAgain = useCallback(() => {
try {
localStorage.setItem(DISMISSED_FOREVER_KEY, "1")
} catch {
// Best-effort; if localStorage fails the user will see the prompt
// again next visit, which is the safe default.
}
setOpen(false)
}, [])
const handleClose = useCallback(() => {
// Session-only dismiss: closing via X or backdrop does NOT persist,
// so the prompt reappears on the next page load. Users who want to
// silence it for longer must use "Not now" (30 d) or "Don't show again".
setOpen(false)
}, [])
if (!open || !platform) return null
return (
<div
role="dialog"
aria-modal="true"
aria-labelledby="pwa-install-title"
className="fixed inset-0 z-[100] flex items-end justify-center bg-black/60 backdrop-blur-sm animate-in fade-in duration-200"
onClick={(e) => {
if (e.target === e.currentTarget) handleClose()
}}
>
<div
className="w-full max-w-md rounded-t-2xl bg-background text-foreground shadow-2xl border-t border-border animate-in slide-in-from-bottom duration-300"
style={{ paddingBottom: "max(1.25rem, env(safe-area-inset-bottom))" }}
>
<div className="relative px-5 pt-5">
<div className="mx-auto mb-3 h-1 w-10 rounded-full bg-border" aria-hidden="true" />
<button
type="button"
onClick={handleClose}
aria-label="Close"
className="absolute right-3 top-3 flex h-8 w-8 items-center justify-center rounded-full text-muted-foreground hover:bg-muted transition-colors"
>
<X className="h-4 w-4" />
</button>
<div className="mb-4 flex items-start gap-3.5">
<div className="flex h-[52px] w-[52px] shrink-0 items-center justify-center rounded-xl bg-muted p-1 shadow-md">
<img src="/icon.svg" alt="ProxMenux Monitor" className="h-full w-full object-contain" />
</div>
<div className="flex-1 min-w-0">
<h3 id="pwa-install-title" className="text-[17px] font-bold leading-tight tracking-tight text-foreground">
Install ProxMenux Monitor
</h3>
<p className="mt-1 text-[13px] leading-snug text-muted-foreground">
{platform === "ios"
? "Add the Monitor to your home screen for quick access."
: "Add the Monitor as an app to launch it like a native application."}
</p>
</div>
</div>
{platform === "ios" ? (
<ol className="mb-4 flex flex-col gap-2" role="list">
<li className="flex items-center gap-3 rounded-xl bg-primary/10 px-3.5 py-3 text-[13.5px] leading-tight">
<span className="flex h-[22px] w-[22px] shrink-0 items-center justify-center rounded-full bg-primary text-[11px] font-bold text-primary-foreground">
1
</span>
<span>
Tap the{" "}
<span className="inline-flex items-center gap-1 font-semibold text-primary">
<Share className="h-4 w-4" aria-hidden="true" />
Share
</span>{" "}
button in the bottom bar
</span>
</li>
<li className="flex items-center gap-3 rounded-xl bg-primary/10 px-3.5 py-3 text-[13.5px] leading-tight">
<span className="flex h-[22px] w-[22px] shrink-0 items-center justify-center rounded-full bg-primary text-[11px] font-bold text-primary-foreground">
2
</span>
<span>
Choose{" "}
<span className="inline-flex items-center gap-1 font-semibold text-primary">
<Plus className="h-4 w-4" aria-hidden="true" />
Add to Home Screen
</span>
</span>
</li>
<li className="flex items-center gap-3 rounded-xl bg-primary/10 px-3.5 py-3 text-[13.5px] leading-tight">
<span className="flex h-[22px] w-[22px] shrink-0 items-center justify-center rounded-full bg-primary text-[11px] font-bold text-primary-foreground">
3
</span>
<span>
Confirm by tapping <b>Add</b> in the top-right
</span>
</li>
</ol>
) : (
<div className="mb-4 rounded-lg border border-border bg-muted/50 px-3.5 py-3 text-[13px] leading-relaxed text-muted-foreground">
Open the browser menu <b className="text-foreground"></b> {" "}
<b className="text-foreground">Add to Home Screen</b> confirm by tapping{" "}
<b className="text-foreground">Install</b>.
</div>
)}
<div className="mt-1 flex flex-col gap-1 border-t border-border pt-3">
<button
type="button"
onClick={handleNotNow}
className="rounded-lg py-2.5 text-center text-[13.5px] font-semibold text-muted-foreground hover:bg-muted transition-colors"
>
Not now
</button>
<button
type="button"
onClick={handleNeverAgain}
className="rounded-lg py-2.5 text-center text-[13.5px] font-semibold text-amber-700 dark:text-amber-500 hover:bg-muted transition-colors"
>
Don&apos;t show again
</button>
</div>
</div>
</div>
</div>
)
}
+21
View File
@@ -0,0 +1,21 @@
"use client"
import { useEffect } from "react"
// Unregister any Service Worker on this origin at mount. A SW here
// interacts badly with mobile battery throttling behind reverse
// proxies. `sw.js` is kept for a future PWA-offline revisit.
export function PwaRegister() {
useEffect(() => {
if (typeof window === "undefined") return
if (!("serviceWorker" in navigator)) return
navigator.serviceWorker
.getRegistrations()
.then((regs) => {
if (regs.length === 0) return
return Promise.all(regs.map((r) => r.unregister()))
})
.catch(() => {})
}, [])
return null
}
+20 -9
View File
@@ -3,10 +3,10 @@
import { useState, useEffect } from "react"
import { Button } from "./ui/button"
import { Dialog, DialogContent, DialogTitle } from "./ui/dialog"
import { X, Sparkles, Thermometer, Activity, HardDrive, Shield, Globe, Cpu, Zap, Sliders, Wrench, RefreshCw, Server, BellOff, Bell } from "lucide-react"
import { X, Sparkles, Thermometer, Activity, HardDrive, Shield, Globe, Cpu, Zap, Sliders, Wrench, RefreshCw, Server, BellOff, Bell, Calendar, DatabaseBackup } from "lucide-react"
import { Checkbox } from "./ui/checkbox"
const APP_VERSION = "1.2.2.1-beta" // Sync with AppImage/package.json
const APP_VERSION = "1.2.4" // Sync with AppImage/package.json
interface ReleaseNote {
date: string
@@ -18,6 +18,21 @@ interface ReleaseNote {
}
export const CHANGELOG: Record<string, ReleaseNote> = {
"1.2.3": {
date: "July 15, 2026",
changes: {
added: [
"Backups integrated in the Monitor — a new first-class section to create, schedule and restore host backups against Local, PBS or Borg destinations from the Web dashboard. Jobs run on a proper systemd timer or attach to an existing PVE vzdump job with retention live-inherited from the parent. Encrypted PBS backups store a paired recovery blob next to each snapshot so a fresh install can always get the key back. After a reboot the tab shows a real-time restore progress card with milestones, per-component status (NVIDIA, Intel GPU tools, Coral, AMD tools), boot sanity warnings and a rollback delta listing anything on the host that wasn't in the backup.",
"Network Flow diagram — a new live topology view on the Network tab showing NICs → host → bridges → LXCs / VMs with animated rx / tx pulses on every internal link, so the operator can see in real time how traffic distributes inside the host and which guests are pulling or pushing data.",
"Physical Disks and Physical Interfaces cards redesigned — clearer per-item presentation on the Storage and Network tabs. USB-NVMe / USB-SATA enclosures reporting removable=0 (ASMedia, JMicron, Realtek, ASM105x) now walk sysfs to detect USB attachment, so the -d snt* pass-through is tried and the drive's real model, serial, temperature, power-on hours and health surface — instead of the bridge's chatter.",
"Richer notifications out of the box — for users not running an AI agent, the templated body now identifies the affected object (which storage, which interface, which container), surfaces the top offenders with an \"…and N more\" tail when the list is long, and preserves the same identity in the recovery message. Users with AI enrichment enabled continue to get their tailored rewrite on top of this improved base.",
],
changed: [
"Redesigned cards across Overview, VM / LXC, Storage and Network — layouts reworked for faster reading and denser, more practical information: key numbers surface at a glance, grouped by relevance, and the responsive grid now behaves cleanly from a phone up to an ultrawide.",
"Health Monitor Thresholds — the Settings panel that controls per-category Warning and Critical levels (CPU, memory, temperature, storage, disks, ...) was reworked with clearer visual grouping and inline hints, so tuning a threshold now takes a couple of clicks instead of scrolling through a wall of numbers.",
],
},
},
"1.2.2": {
date: "May 31, 2026",
changes: {
@@ -216,17 +231,13 @@ export const CHANGELOG: Record<string, ReleaseNote> = {
}
const CURRENT_VERSION_FEATURES = [
{
icon: <Activity className="h-5 w-5" />,
text: "Header Critical badge now respects dismissals (#228) - Permanently silencing every critical alert in a category used to leave the badge stuck on Critical even though the popup correctly reported 0 critical. The rollup that drives /api/system-info now runs a dismiss-aware pass over every category, so the badge, the popup and any API consumer all see the same view",
},
{
icon: <RefreshCw className="h-5 w-5" />,
text: "Auto-reconcile of stale alerts - Errors for resources that no longer exist now auto-clear within the regular cleanup cycle. New cases: a PVE storage removed via pvesm, an NFS/CIFS share whose mount target is no longer in /proc/mounts (the lazy-umount case reported in the field), and LXC mount-capacity alerts whose CT has been deleted",
text: "One-click host update from the Health Monitor — new Update Now button in the System Updates section runs the Proxmox update flow in an in-dashboard terminal, without leaving the browser.",
},
{
icon: <Bell className="h-5 w-5" />,
text: "Notification Send Test buttons unified (#226) - All five channel Send Test buttons (Telegram, Gotify, Discord, Email, Apprise) now sit on the left side and carry their channel's brand colour with white text, instead of Apprise being the right-aligned cyan outlier",
icon: <Sparkles className="h-5 w-5" />,
text: "In-app Install prompt for mobile — first-time visitors on Android and iOS Safari now see a bottom-sheet with clear steps for adding the Monitor to their home screen as a PWA.",
},
]
@@ -0,0 +1,713 @@
"use client"
// Live inline card + detail modal for the post-boot restore.
//
// apply_cluster_postboot.sh writes /var/lib/proxmenux/restore-state.json
// as it works through the milestones (apply cluster config, initramfs,
// grub, per-component reinstalls, sanity check, finalize). The Flask
// endpoints /api/host-backups/restore/{status,dismiss,history,log}
// expose that state to this component. While the restore is running we
// poll every 2s; once it's terminal (complete|failed) we back off to
// 30s so the card can still be re-opened as a summary. Once the
// operator hits Dismiss the card collapses and the History button
// keeps the run browsable.
import { useMemo, useState } from "react"
import useSWR from "swr"
import { Card, CardContent, CardHeader, CardTitle } from "./ui/card"
import { Button } from "./ui/button"
import { Badge } from "./ui/badge"
import {
Dialog,
DialogContent,
DialogHeader,
DialogTitle,
DialogDescription,
DialogFooter,
} from "./ui/dialog"
import { ScrollArea } from "./ui/scroll-area"
import {
Loader2,
CheckCircle2,
XCircle,
AlertTriangle,
History,
RotateCcw,
ChevronRight,
Cpu,
FileText,
ArrowDownAZ,
Filter,
} from "lucide-react"
import { fetchApi } from "../lib/api-config"
// ── Shape contracts with the backend ──────────────────────────
interface RestoreComponent {
name: string
status: "installing" | "ok" | "failed"
log: string
exit_code?: string
}
interface RestoreSummary {
hostname: string
guests: string
stubs: string
stale_nodes: string
components: string
duration: string
}
interface RestoreRollback {
vms_to_remove?: string[]
lxcs_to_remove?: string[]
components_to_uninstall?: string[]
}
interface DataPoolsImport {
ok: string[]
forced: string[]
partial: string[]
missing: string[]
failed: string[]
finished_at?: string
log_path?: string
}
interface RestoreState {
status: "running" | "complete" | "failed"
started_at: string
finished_at: string | null
current_step: string
steps_done: number
steps_total: number
log_path: string
components: RestoreComponent[]
rollback_delta: RestoreRollback
sanity_warnings: string[]
summary: RestoreSummary | null
acknowledged: boolean
duration?: string
data_pools_import?: DataPoolsImport
}
interface HistoryEntry {
file: string
mtime: number
status: string
started_at: string | null
finished_at: string | null
duration: string | null
}
const fetcher = (url: string) => fetchApi(url)
const COMPONENT_LABEL: Record<string, string> = {
nvidia_driver: "NVIDIA driver",
amdgpu_top: "amdgpu_top",
intel_gpu_tools: "Intel GPU tools",
coral_driver: "Google Coral TPU driver",
}
const formatComponent = (name: string) => COMPONENT_LABEL[name] ?? name
const formatIso = (iso: string | null | undefined) => {
if (!iso) return "—"
try {
return new Date(iso).toLocaleString()
} catch {
return iso
}
}
const formatRelative = (iso: string) => {
try {
const then = new Date(iso).getTime()
const now = Date.now()
const diff = Math.max(0, Math.round((now - then) / 1000))
if (diff < 60) return `${diff}s ago`
if (diff < 3600) return `${Math.round(diff / 60)}m ago`
if (diff < 86400) return `${Math.round(diff / 3600)}h ago`
return `${Math.round(diff / 86400)}d ago`
} catch {
return iso
}
}
// Rough time-remaining estimate derived from steps_done + elapsed.
// Best-effort: at step 0 there's no data yet, so it returns
// "estimating time…". After the run is terminal, "—". The output is
// a full phrase so the caller doesn't have to add suffix words that
// only make sense on some branches.
const computeEta = (state: RestoreState): string => {
if (state.status !== "running") return "—"
if (!state.steps_done || state.steps_done <= 0) return "estimating time…"
const elapsedSec = Math.max(1, Math.round((Date.now() - new Date(state.started_at).getTime()) / 1000))
const perStep = elapsedSec / state.steps_done
const remaining = Math.max(0, state.steps_total - state.steps_done)
const eta = Math.round(perStep * remaining)
if (eta < 60) return `~${eta}s left`
if (eta < 3600) return `~${Math.round(eta / 60)}m left`
return `~${Math.round(eta / 3600)}h left`
}
// ── Small building blocks ─────────────────────────────────────
const StatusBadge: React.FC<{ status: string }> = ({ status }) => {
if (status === "running")
return (
<Badge className="bg-blue-500/10 border-blue-500/40 text-blue-300 gap-1">
<Loader2 className="h-3 w-3 animate-spin" />
Restore in progress
</Badge>
)
if (status === "complete")
return (
<Badge className="bg-emerald-500/10 border-emerald-500/40 text-emerald-400 gap-1">
<CheckCircle2 className="h-3 w-3" />
Restore complete
</Badge>
)
if (status === "failed")
return (
<Badge className="bg-red-500/10 border-red-500/40 text-red-400 gap-1">
<XCircle className="h-3 w-3" />
Restore failed
</Badge>
)
return <Badge variant="outline">{status}</Badge>
}
const ComponentStatusIcon: React.FC<{ status: string }> = ({ status }) => {
if (status === "installing")
return <Loader2 className="h-3.5 w-3.5 animate-spin text-blue-400" />
if (status === "ok")
return <CheckCircle2 className="h-3.5 w-3.5 text-emerald-400" />
return <XCircle className="h-3.5 w-3.5 text-red-400" />
}
// ── Log viewer ────────────────────────────────────────────────
const LogViewer: React.FC<{ path: string | null; historyOnly?: boolean }> = ({ path, historyOnly }) => {
const [filter, setFilter] = useState<"all" | "issues">("all")
const swrKey = path
? `/api/host-backups/restore/log?filter=${filter}&tail=600${historyOnly ? `&path=${encodeURIComponent(path)}` : ""}`
: null
const { data, isLoading } = useSWR<{ lines: string[]; total_lines: number; path: string | null }>(
swrKey,
fetcher,
{ refreshInterval: historyOnly ? 0 : 4000 },
)
return (
<div className="space-y-2">
<div className="flex items-center justify-between text-xs">
<div className="flex items-center gap-1 text-muted-foreground">
<FileText className="h-3.5 w-3.5" />
{path ?? "no log yet"}
</div>
<div className="flex items-center gap-1">
<Button
size="sm"
variant={filter === "all" ? "default" : "outline"}
className="h-6 px-2 text-xs"
onClick={() => setFilter("all")}
>
<ArrowDownAZ className="h-3 w-3 mr-1" />
Full
</Button>
<Button
size="sm"
variant={filter === "issues" ? "default" : "outline"}
className="h-6 px-2 text-xs"
onClick={() => setFilter("issues")}
>
<Filter className="h-3 w-3 mr-1" />
Issues only
</Button>
</div>
</div>
<ScrollArea className="h-72 rounded-md border border-border bg-black/40">
<pre className="p-3 text-xs text-muted-foreground whitespace-pre-wrap font-mono leading-relaxed">
{isLoading ? "Loading…" : (data?.lines?.join("\n") || "(no output)")}
</pre>
</ScrollArea>
</div>
)
}
// ── Rollback delta widget ─────────────────────────────────────
const RollbackDelta: React.FC<{ delta: RestoreRollback | undefined }> = ({ delta }) => {
const vms = delta?.vms_to_remove ?? []
const lxcs = delta?.lxcs_to_remove ?? []
const comps = delta?.components_to_uninstall ?? []
if (!vms.length && !lxcs.length && !comps.length) {
return (
<div className="text-xs text-muted-foreground">
No entries exist on this host that weren't in the restored backup.
</div>
)
}
const Row: React.FC<{ label: string; items: string[]; cmd: (id: string) => string }> = ({ label, items, cmd }) =>
items.length === 0 ? null : (
<div className="space-y-1">
<div className="text-xs font-medium text-muted-foreground">{label}</div>
<div className="flex flex-wrap gap-1.5">
{items.map((id) => (
<Badge key={id} variant="outline" className="font-mono text-xs">
{id}
</Badge>
))}
</div>
{items.length > 0 && (
<details className="text-xs">
<summary className="cursor-pointer text-muted-foreground hover:text-foreground">
Show manual cleanup commands
</summary>
<pre className="mt-1 p-2 rounded-md bg-black/40 text-xs text-muted-foreground font-mono">
{items.map(cmd).join("\n")}
</pre>
</details>
)}
</div>
)
return (
<div className="space-y-3">
<div className="text-xs text-muted-foreground">
These entries exist on this host but were NOT in the restored backup. Review before removing.
</div>
<Row
label="VMs created after the backup"
items={vms}
cmd={(id) => `qm stop ${id} 2>/dev/null; qm destroy ${id} --purge`}
/>
<Row
label="LXCs created after the backup"
items={lxcs}
cmd={(id) => `pct stop ${id} 2>/dev/null; pct destroy ${id} --purge`}
/>
<Row
label="Components installed after the backup"
items={comps}
cmd={(name) => `# uninstall ${name} manually via ProxMenux → Hardware & GPU`}
/>
</div>
)
}
// ── Detail modal ──────────────────────────────────────────────
const RestoreDetailModal: React.FC<{
open: boolean
onClose: () => void
state: RestoreState
historyMode?: boolean
}> = ({ open, onClose, state, historyMode }) => {
const progressPct = state.steps_total > 0 ? Math.round((state.steps_done / state.steps_total) * 100) : 0
return (
<Dialog open={open} onOpenChange={(v) => !v && onClose()}>
<DialogContent className="max-w-3xl">
<DialogHeader>
<DialogTitle className="flex items-center gap-2">
<RotateCcw className="h-5 w-5 text-blue-500" />
Post-restore progress
<StatusBadge status={state.status} />
</DialogTitle>
<DialogDescription>
Started {formatIso(state.started_at)}
{state.finished_at ? ` · finished ${formatIso(state.finished_at)}` : ""}
{state.summary?.duration ? ` · ${state.summary.duration}` : ""}
</DialogDescription>
</DialogHeader>
<div className="space-y-4">
<div className="space-y-1">
<div className="flex justify-between text-xs text-muted-foreground">
<span>{state.current_step || "—"}</span>
<span>
{state.steps_done}/{state.steps_total} steps
{state.status === "running" && ` · ${computeEta(state)}`}
</span>
</div>
<div className="h-2 rounded-full bg-muted overflow-hidden">
<div
className={`h-full transition-all duration-500 ${
state.status === "failed" ? "bg-red-500" : state.status === "complete" ? "bg-emerald-500" : "bg-blue-500"
}`}
style={{ width: `${progressPct}%` }}
/>
</div>
</div>
{state.components.length > 0 && (
<div className="space-y-2">
<div className="text-sm font-medium flex items-center gap-2">
<Cpu className="h-4 w-4" />
Components
</div>
<div className="space-y-1.5">
{state.components.map((c) => (
<div
key={c.name}
className="flex items-center justify-between rounded-md border border-border bg-muted/30 px-3 py-2 text-xs"
>
<div className="flex items-center gap-2">
<ComponentStatusIcon status={c.status} />
<span className="font-medium">{formatComponent(c.name)}</span>
<span className="text-muted-foreground">{c.status}</span>
{c.exit_code && <span className="text-red-400">exit {c.exit_code}</span>}
</div>
{c.log && <span className="text-muted-foreground font-mono">{c.log}</span>}
</div>
))}
</div>
</div>
)}
{state.sanity_warnings.length > 0 && (
<div className="space-y-2">
<div className="text-sm font-medium flex items-center gap-2 text-amber-400">
<AlertTriangle className="h-4 w-4" />
Boot sanity warnings
</div>
<ul className="list-disc list-inside text-xs text-muted-foreground space-y-1">
{state.sanity_warnings.map((w) => (
<li key={w}>{w}</li>
))}
</ul>
</div>
)}
{state.data_pools_import && <DataPoolsBlock section={state.data_pools_import} />}
<div className="space-y-2">
<div className="text-sm font-medium">Rollback delta</div>
<RollbackDelta delta={state.rollback_delta} />
</div>
<div className="space-y-2">
<div className="text-sm font-medium">Log</div>
<LogViewer path={state.log_path} historyOnly={historyMode} />
</div>
</div>
<DialogFooter>
<Button variant="outline" onClick={onClose}>
Close
</Button>
</DialogFooter>
</DialogContent>
</Dialog>
)
}
// Rendered inside RestoreDetailModal — one row per outcome category
// (imported / forced / partial skip / missing skip / failed).
const DataPoolsBlock: React.FC<{ section: DataPoolsImport }> = ({ section }) => {
const total =
section.ok.length +
section.forced.length +
section.partial.length +
section.missing.length +
section.failed.length
if (total === 0) return null
const Row: React.FC<{
label: string
tone: "ok" | "warn" | "info" | "error"
items: string[]
help?: string
}> = ({ label, tone, items, help }) => {
if (items.length === 0) return null
const toneClass =
tone === "ok"
? "text-emerald-400"
: tone === "warn"
? "text-amber-400"
: tone === "error"
? "text-red-400"
: "text-blue-400"
return (
<div className="rounded-md border border-border bg-muted/30 px-3 py-2 text-xs">
<div className={`font-medium ${toneClass} flex items-center gap-2`}>
{tone === "ok" && <CheckCircle2 className="h-3.5 w-3.5" />}
{tone === "warn" && <AlertTriangle className="h-3.5 w-3.5" />}
{tone === "error" && <XCircle className="h-3.5 w-3.5" />}
{tone === "info" && <CheckCircle2 className="h-3.5 w-3.5" />}
<span>{label}</span>
<span className="text-muted-foreground">({items.length})</span>
</div>
<div className="mt-1 font-mono text-muted-foreground break-all">{items.join(", ")}</div>
{help && <div className="mt-1 text-muted-foreground">{help}</div>}
</div>
)
}
return (
<div className="space-y-2">
<div className="text-sm font-medium flex items-center gap-2">
<Cpu className="h-4 w-4" />
ZFS data pools auto-import
</div>
<div className="space-y-1.5">
<Row label="Imported" tone="ok" items={section.ok} />
<Row
label="Imported (forced, foreign hostid)"
tone="info"
items={section.forced}
help="New hostid grabbed onto the pool label — next boot imports clean."
/>
<Row
label="Skipped (some disks missing)"
tone="warn"
items={section.partial}
help="Some vdev disks weren't found by /dev/disk/by-id. Pool NOT imported to avoid a degraded auto-import. Fix the disks or import manually with zpool import."
/>
<Row
label="Skipped (no disks present)"
tone="warn"
items={section.missing}
help="None of the pool's disks are on this host. Move the disks over or import from a different host."
/>
<Row
label="Import failed"
tone="error"
items={section.failed}
help="ZFS rejected the import even with -f. Inspect with `zpool import` and the log below."
/>
</div>
{section.log_path && (
<div className="text-xs text-muted-foreground font-mono">Log: {section.log_path}</div>
)}
</div>
)
}
// ── History browser modal ─────────────────────────────────────
const RestoreHistoryModal: React.FC<{ open: boolean; onClose: () => void }> = ({ open, onClose }) => {
const { data } = useSWR<{ entries: HistoryEntry[] }>(open ? "/api/host-backups/restore/history" : null, fetcher)
const [detailFile, setDetailFile] = useState<string | null>(null)
const { data: detailResp } = useSWR<{ state: RestoreState }>(
detailFile ? `/api/host-backups/restore/history?file=${encodeURIComponent(detailFile)}` : null,
fetcher,
)
return (
<>
<Dialog open={open} onOpenChange={(v) => !v && onClose()}>
<DialogContent className="max-w-2xl">
<DialogHeader>
<DialogTitle className="flex items-center gap-2">
<History className="h-5 w-5" />
Past restores
</DialogTitle>
<DialogDescription>
Restores archived by the post-boot dispatcher. The latest 20 are kept.
</DialogDescription>
</DialogHeader>
<ScrollArea className="h-96">
<div className="space-y-1.5">
{(data?.entries ?? []).length === 0 ? (
<div className="text-sm text-muted-foreground py-6 text-center">No past restores recorded.</div>
) : (
(data?.entries ?? []).map((e) => (
<button
key={e.file}
onClick={() => setDetailFile(e.file)}
className="w-full flex items-center justify-between rounded-md border border-border bg-muted/30 hover:bg-muted px-3 py-2 text-xs text-left"
>
<div className="flex items-center gap-2">
<StatusBadge status={e.status} />
<span className="text-muted-foreground">
{e.started_at ? formatIso(e.started_at) : formatIso(new Date(e.mtime * 1000).toISOString())}
</span>
{e.duration && <span className="text-muted-foreground">· {e.duration}</span>}
</div>
<ChevronRight className="h-4 w-4 text-muted-foreground" />
</button>
))
)}
</div>
</ScrollArea>
<DialogFooter>
<Button variant="outline" onClick={onClose}>
Close
</Button>
</DialogFooter>
</DialogContent>
</Dialog>
{detailFile && detailResp?.state && (
<RestoreDetailModal
open={!!detailFile}
onClose={() => setDetailFile(null)}
state={detailResp.state}
historyMode
/>
)}
</>
)
}
// ── Main inline card ──────────────────────────────────────────
export const RestoreProgressCard: React.FC = () => {
const { data, mutate } = useSWR<{ state: RestoreState | null }>(
"/api/host-backups/restore/status",
fetcher,
{
refreshInterval: (latest) => (latest?.state?.status === "running" ? 2000 : 30000),
revalidateOnFocus: true,
},
)
const [detailOpen, setDetailOpen] = useState(false)
const [historyOpen, setHistoryOpen] = useState(false)
const [dismissing, setDismissing] = useState(false)
const state = data?.state ?? null
const progressPct = useMemo(() => {
if (!state || state.steps_total <= 0) return 0
return Math.round((state.steps_done / state.steps_total) * 100)
}, [state])
const dismiss = async () => {
if (!state) return
setDismissing(true)
try {
await fetchApi("/api/host-backups/restore/dismiss", { method: "POST" })
await mutate()
} finally {
setDismissing(false)
}
}
// Hidden entirely when: no restore run has ever happened, OR the
// last run is terminal AND acknowledged. History button is still
// reachable from the main card header (rendered elsewhere).
if (!state) return null
if (state.status !== "running" && state.acknowledged) {
return (
<div className="flex justify-end">
<Button variant="ghost" size="sm" onClick={() => setHistoryOpen(true)}>
<History className="h-3.5 w-3.5 mr-1" />
Past restores
</Button>
<RestoreHistoryModal open={historyOpen} onClose={() => setHistoryOpen(false)} />
</div>
)
}
const hasWarnings = state.sanity_warnings.length > 0
const pools = state.data_pools_import
const poolCount =
(pools?.ok.length ?? 0) +
(pools?.forced.length ?? 0) +
(pools?.partial.length ?? 0) +
(pools?.missing.length ?? 0) +
(pools?.failed.length ?? 0)
const poolWarnings = (pools?.partial.length ?? 0) + (pools?.missing.length ?? 0) + (pools?.failed.length ?? 0)
const barColor =
state.status === "failed" ? "bg-red-500" : state.status === "complete" ? "bg-emerald-500" : "bg-blue-500"
return (
<>
<Card className="bg-card border-border">
<CardHeader className="pb-3">
<div className="flex flex-wrap items-center justify-between gap-2">
<CardTitle className="text-base font-semibold flex items-center gap-2">
<RotateCcw
className={`h-5 w-5 ${state.status === "running" ? "text-blue-500 animate-spin" : "text-blue-500"}`}
/>
Post-restore progress
<StatusBadge status={state.status} />
{hasWarnings && (
<Badge variant="outline" className="text-amber-400 border-amber-500/40 bg-amber-500/10 gap-1">
<AlertTriangle className="h-3 w-3" />
{state.sanity_warnings.length} boot warning{state.sanity_warnings.length === 1 ? "" : "s"}
</Badge>
)}
{poolCount > 0 && (
<Badge
variant="outline"
className={
poolWarnings > 0
? "text-amber-400 border-amber-500/40 bg-amber-500/10 gap-1"
: "text-emerald-400 border-emerald-500/40 bg-emerald-500/10 gap-1"
}
>
<Cpu className="h-3 w-3" />
{poolCount} ZFS pool{poolCount === 1 ? "" : "s"}
{poolWarnings > 0 && ` · ${poolWarnings} need attention`}
</Badge>
)}
</CardTitle>
<div className="flex items-center gap-2">
<Button size="sm" variant="outline" onClick={() => setDetailOpen(true)}>
Details
</Button>
<Button size="sm" variant="ghost" onClick={() => setHistoryOpen(true)}>
<History className="h-3.5 w-3.5 mr-1" />
History
</Button>
{state.status !== "running" && (
<Button size="sm" onClick={dismiss} disabled={dismissing}>
{dismissing ? <Loader2 className="h-3.5 w-3.5 animate-spin" /> : "Dismiss"}
</Button>
)}
</div>
</div>
</CardHeader>
<CardContent className="space-y-3">
<div className="space-y-1">
<div className="flex justify-between text-xs text-muted-foreground">
<span className="truncate">
{state.current_step || "—"} · started {formatRelative(state.started_at)}
</span>
<span>
{state.steps_done}/{state.steps_total} steps
{state.status === "running" && ` · ${computeEta(state)}`}
{state.summary?.duration && state.status !== "running" && ` · ${state.summary.duration}`}
</span>
</div>
<div className="h-2 rounded-full bg-muted overflow-hidden">
<div className={`h-full transition-all duration-500 ${barColor}`} style={{ width: `${progressPct}%` }} />
</div>
</div>
{state.summary && (
<div className="grid grid-cols-2 md:grid-cols-4 gap-2 text-xs">
<div className="rounded-md border border-border bg-muted/30 px-2 py-1.5">
<div className="text-muted-foreground">Guests</div>
<div className="font-medium">{state.summary.guests}</div>
</div>
<div className="rounded-md border border-border bg-muted/30 px-2 py-1.5">
<div className="text-muted-foreground">Bind-mount stubs</div>
<div className="font-medium">{state.summary.stubs}</div>
</div>
<div className="rounded-md border border-border bg-muted/30 px-2 py-1.5">
<div className="text-muted-foreground">Stale nodes cleaned</div>
<div className="font-medium">{state.summary.stale_nodes}</div>
</div>
<div className="rounded-md border border-border bg-muted/30 px-2 py-1.5">
<div className="text-muted-foreground">Components</div>
<div className="font-medium">{state.summary.components}</div>
</div>
</div>
)}
</CardContent>
</Card>
<RestoreDetailModal open={detailOpen} onClose={() => setDetailOpen(false)} state={state} />
<RestoreHistoryModal open={historyOpen} onClose={() => setHistoryOpen(false)} />
</>
)
}
export default RestoreProgressCard
@@ -49,6 +49,12 @@ interface ScriptTerminalModalProps {
description: string
scriptName?: string
params?: Record<string, string>
// Optional callback fired when the script's WebSocket closes
// (script_runner sends an exit code and then closes). Lets the
// parent auto-dismiss the modal — used by host-backup's Restore
// flow so "Press Enter to close" in the bash script actually
// closes the modal without an extra click. Other callers ignore.
onComplete?: () => void
}
export function ScriptTerminalModal({
@@ -58,6 +64,7 @@ export function ScriptTerminalModal({
title,
description,
params = { EXECUTION_MODE: "web" },
onComplete,
}: ScriptTerminalModalProps) {
const termRef = useRef<any>(null)
const wsRef = useRef<WebSocket | null>(null)
@@ -95,6 +102,13 @@ export function ScriptTerminalModal({
paramsRef.current = params
}, [params])
// Same trick for onComplete — we want the latest callback inside
// the ws.onclose handler without re-running the connection effect.
const onCompleteRef = useRef<(() => void) | undefined>(undefined)
useEffect(() => {
onCompleteRef.current = onComplete
}, [onComplete])
const attemptReconnect = useCallback(() => {
if (!isOpen || isComplete || reconnectAttemptsRef.current >= 3) {
return
@@ -195,6 +209,7 @@ const initMessage = {
reconnectTimeoutRef.current = setTimeout(attemptReconnect, 2000)
} else {
setIsComplete(true)
onCompleteRef.current?.()
}
}
}
@@ -284,6 +299,11 @@ const initMessage = {
setTimeout(() => {
if (fitAddonRef.current && termRef.current) {
fitAddonRef.current.fit()
// Send the keyboard to the xterm instance so any --yesno /
// --menu the script opens receives Enter / arrow keys
// directly. Without this the modal's Close button (or any
// other focusable descendant) intercepts the keystrokes.
termRef.current.focus()
}
}, 100)
@@ -382,6 +402,7 @@ const initMessage = {
if (!isComplete) {
setIsComplete(true)
onCompleteRef.current?.()
}
}
@@ -666,6 +687,13 @@ const initMessage = {
}}
onInteractOutside={(e) => e.preventDefault()}
onEscapeKeyDown={(e) => e.preventDefault()}
// Radix defaults to focusing the first focusable descendant
// when a Dialog opens — that used to land on the Close button
// and every Enter/Space keystroke intended for the shell
// dialogs INSIDE the terminal would close the modal instead.
// Preempting the auto-focus lets the xterm focus() call in
// initializeTerminal below own the keyboard from the start.
onOpenAutoFocus={(e) => e.preventDefault()}
hideClose
>
<DialogTitle className="sr-only">{title}</DialogTitle>
+1 -1
View File
@@ -108,7 +108,7 @@ export function StorageMetrics() {
return (
<div className="space-y-6">
{/* Storage Overview Cards */}
<div className="grid grid-cols-2 lg:grid-cols-4 gap-3 lg:gap-6">
<div className="grid grid-cols-2 xl:grid-cols-4 gap-3 xl:gap-6">
<Card className="bg-card border-border">
<CardHeader className="flex flex-row items-center justify-between space-y-0 pb-2">
<CardTitle className="text-sm font-medium text-muted-foreground">Total Storage</CardTitle>
+375 -395
View File
@@ -2,14 +2,16 @@
import { useEffect, useState } from "react"
import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card"
import { HardDrive, Database, AlertTriangle, CheckCircle2, XCircle, Square, Thermometer, Archive, Info, Clock, Usb, Server, Activity, FileText, Play, Loader2, Download, Plus, Trash2, Settings } from "lucide-react"
import { HardDrive, Database, AlertTriangle, CheckCircle2, XCircle, Square, Thermometer, Archive, Info, Clock, Usb, Server, Activity, FileText, Play, Loader2, Download, Plus, Trash2, Settings, Power } from "lucide-react"
import { Badge } from "@/components/ui/badge"
import { Progress } from "@/components/ui/progress"
import { Dialog, DialogContent, DialogDescription, DialogHeader, DialogTitle } from "@/components/ui/dialog"
import { Button } from "@/components/ui/button"
import { fetchApi } from "../lib/api-config"
import { formatStorage as sharedFormatStorage } from "../lib/utils"
import { DiskTemperatureDetailModal } from "./disk-temperature-detail-modal"
import { DiskTemperatureCard } from "./disk-temperature-card"
import { getDiskType as resolveDiskType } from "../lib/disk-type"
import {
useDiskTempThresholds,
loadDiskTempThresholds,
@@ -22,6 +24,11 @@ interface DiskInfo {
size?: number // Changed from string to number (KB) for formatMemory()
size_formatted?: string // Added formatted size string for display
temperature: number
// True when the temperature poller's last smartctl exited with
// "device is in standby". The UI uses this to render a Standby
// badge AND to suppress the (stale) temperature value, so the
// operator understands the graph is frozen on purpose — issue #232.
standby?: boolean
health: string
power_on_hours?: number
smart_status?: string
@@ -141,17 +148,9 @@ interface RemoteMountsData {
error?: string
}
const formatStorage = (sizeInGB: number): string => {
if (sizeInGB < 1) {
// Less than 1 GB, show in MB
return `${(sizeInGB * 1024).toFixed(1)} MB`
} else if (sizeInGB > 999) {
return `${(sizeInGB / 1024).toFixed(2)} TB`
} else {
// Between 1 and 999 GB, show in GB
return `${sizeInGB.toFixed(2)} GB`
}
}
// Re-exported under the local name so the rest of this large file
// stays untouched. Single source of truth lives in lib/utils.ts.
const formatStorage = sharedFormatStorage
// Translate the short ATA/SCSI error codes that appear inside `{ ... }`
// in a raw kernel observation (e.g. `error: { IDNF }`) into a one-line
@@ -279,6 +278,301 @@ export function StorageOverview() {
}
}
// Tiny coloured dot that prefixes status / counter values. Adds
// accessibility-friendly redundancy (colour + position) to fields
// that today rely on colour alone, so an "all OK" disk reads as
// visually quiet and a degraded one as immediately noisy.
//
// Colour mapping:
// - "ok" → green (passed / 0 errors)
// - "warn" → amber (1+ errors but not critical)
// - "fail" → red (failed / many errors)
const StatusDot = ({ tone }: { tone: "ok" | "warn" | "fail" }) => {
const cls =
tone === "ok" ? "bg-green-500" : tone === "warn" ? "bg-yellow-500" : "bg-red-500"
return (
<span
className={`inline-block h-2 w-2 rounded-full shrink-0 ${cls}`}
aria-hidden
/>
)
}
// Decide the tone for a counter where 0 is healthy. The "warn" /
// "fail" cutoffs are conservative — even a single reallocated
// sector is worth amber attention, and double digits start hinting
// at progressive failure (red).
const counterTone = (n: number | null | undefined): "ok" | "warn" | "fail" => {
if (!n || n <= 0) return "ok"
if (n < 10) return "warn"
return "fail"
}
const smartStatusTone = (s: string | undefined): "ok" | "warn" | "fail" => {
const v = (s || "").toLowerCase()
if (v === "passed" || v === "ok") return "ok"
if (v === "failed") return "fail"
return "warn"
}
// Renders either the live temperature or a "Standby" badge for a
// spun-down drive. Centralised here because the same pattern shows up
// in 4 different disk-list views (system / data / pool / other) and we
// want them all to behave identically — issue #232 fix.
const renderDiskTempOrStandby = (disk: DiskInfo) => {
if (disk.standby) {
return (
<Badge
className="bg-blue-500/10 text-blue-300 border-blue-500/30 gap-1"
title="Drive is in standby — smartctl skipped to keep it spun down"
>
<Power className="h-3 w-3" />
Standby
</Badge>
)
}
if (disk.temperature > 0) {
return (
<div className="flex items-center gap-1">
<Thermometer className={`h-4 w-4 ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`} />
<span className={`text-sm font-medium ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`}>
{disk.temperature}°C
</span>
</div>
)
}
return null
}
// ──────────────────────────────────────────────────────────────────
// Disk card layout (ghosthvj-style)
// ──────────────────────────────────────────────────────────────────
// Single card per disk, grid-arranged at the parent level. Two-line
// header (identity + status / size + temp), separator, vertical
// key→value stat list with WEAR LEVEL bar when available, and a
// footer with serial + "Ver detalles →". Replaces the previous
// duplicated mobile/desktop full-width rows.
const renderDiskCardV2 = (disk: DiskInfo) => {
const type = getDiskTypeBadge(disk.name, disk.rotation_rate)
// Pick the most relevant wear metric for the device class.
// NVMe uses `percentage_used` (0 = fresh, 100 = TBW spent).
// SSD may expose `media_wearout_indicator` (decreasing 100→0) or
// `ssd_life_left` (decreasing 100→0). Normalise both to a
// "percentage spent" so the bar always fills LEFT to RIGHT as the
// drive ages — visually consistent across vendors.
let wearPct: number | null = null
if (typeof disk.percentage_used === "number") wearPct = disk.percentage_used
else if (typeof disk.ssd_life_left === "number") wearPct = 100 - disk.ssd_life_left
else if (typeof disk.media_wearout_indicator === "number")
wearPct = 100 - disk.media_wearout_indicator
// Wear bar always uses the same blue as the modal's wear visual,
// even when the wear is high — the colour is the SECTION colour,
// not a severity signal. The percentage value itself (and the
// surrounding stats) already communicate health via the dot
// colours, so flipping the bar to amber/red here would just
// double-encode the same thing and break visual consistency
// with the detail modal.
const wearColor = wearPct === null ? "" : "bg-blue-500"
const cleanSerial = (disk.serial || "").replace(/\\x[0-9a-fA-F]{2}/g, "")
return (
<div
key={disk.name}
className="border border-white/10 rounded-lg p-5 cursor-pointer bg-card hover:bg-white/5 transition-colors flex flex-col"
onClick={() => handleDiskClick(disk)}
>
{/* Header line 1: identity + SMART status (right). */}
<div className="flex items-start justify-between gap-3">
<div className="flex items-center gap-2 flex-wrap min-w-0">
<h3 className="font-mono font-bold text-base break-all">/dev/{disk.name}</h3>
<Badge className={type.className}>{type.label}</Badge>
{disk.is_system_disk && (
<Badge className="bg-orange-500/10 text-orange-500 border-orange-500/20 gap-1">
<Server className="h-3 w-3" />
System
</Badge>
)}
{disk.connection_type === "usb" && (
<Badge className="bg-orange-500/10 text-orange-400 border-orange-500/20 gap-1">
<Usb className="h-3 w-3" />
USB
</Badge>
)}
</div>
{disk.smart_status && disk.smart_status !== "unknown" && (
<span
className={`flex items-center gap-1.5 text-sm font-semibold uppercase tracking-wide shrink-0 ${
smartStatusTone(disk.smart_status) === "ok"
? "text-green-500"
: smartStatusTone(disk.smart_status) === "fail"
? "text-red-500"
: "text-muted-foreground"
}`}
>
<StatusDot tone={smartStatusTone(disk.smart_status)} />
{disk.smart_status}
</span>
)}
</div>
{/* Header line 2: size + temperature/standby. */}
<div className="flex items-center justify-between gap-3 mt-1">
<span className="text-sm text-muted-foreground">{disk.size_formatted}</span>
{disk.standby ? (
<Badge
className="bg-blue-500/10 text-blue-300 border-blue-500/30 gap-1"
title="Drive is in standby — smartctl skipped to keep it spun down"
>
<Power className="h-3 w-3" />
Standby
</Badge>
) : disk.temperature > 0 ? (
<span
className={`text-base font-semibold ${getTempColor(
disk.temperature,
disk.name,
disk.rotation_rate,
)}`}
>
{disk.temperature}°C
</span>
) : null}
</div>
{/* I/O errors banner (preserved from the previous design). */}
{disk.io_errors && disk.io_errors.count > 0 && (
<div
className={`mt-3 flex items-start gap-2 p-2 rounded text-xs ${
disk.io_errors.severity === "CRITICAL"
? "bg-red-500/10 text-red-400 border border-red-500/20"
: "bg-yellow-500/10 text-yellow-400 border border-yellow-500/20"
}`}
>
<AlertTriangle className="h-3.5 w-3.5 flex-shrink-0 mt-0.5" />
<span>
{disk.io_errors.error_type === "filesystem"
? "Filesystem corruption detected"
: `${disk.io_errors.count} I/O error${
disk.io_errors.count !== 1 ? "s" : ""
} in 5 min`}
</span>
</div>
)}
{/* Separator. */}
<div className="border-t border-border/60 my-3" />
{/* Stats: vertical keyvalue list. Each row matches the
"uppercase label left · value right" pattern from ghosthvj. */}
<div className="space-y-2 text-sm">
{disk.model && disk.model !== "Unknown" && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground shrink-0">
Model
</span>
<span className="font-medium text-right truncate font-mono text-xs">{disk.model}</span>
</div>
)}
{wearPct !== null && (
<div>
<div className="flex items-baseline justify-between gap-3 mb-1">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground">
Wear Level
</span>
<span className="font-medium">{wearPct}%</span>
</div>
<div className="h-1.5 rounded-full bg-muted/40 overflow-hidden">
<div
className={`h-full ${wearColor}`}
style={{ width: `${Math.min(100, Math.max(0, wearPct))}%` }}
/>
</div>
</div>
)}
{disk.power_cycles !== undefined && disk.power_cycles > 0 && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground">
Power Cycles
</span>
<span className="font-medium">{disk.power_cycles.toLocaleString()}</span>
</div>
)}
{disk.power_on_hours !== undefined && disk.power_on_hours > 0 && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground">
Power On
</span>
<span className="font-medium">{formatHours(disk.power_on_hours)}</span>
</div>
)}
{disk.crc_errors !== undefined && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground">
CRC Errors
</span>
<span
className={`font-medium flex items-center gap-1.5 ${
counterTone(disk.crc_errors) === "ok"
? "text-green-500"
: counterTone(disk.crc_errors) === "warn"
? "text-yellow-500"
: "text-red-500"
}`}
>
<StatusDot tone={counterTone(disk.crc_errors)} />
{disk.crc_errors}
</span>
</div>
)}
{/* Reallocated only meaningful on rotating disks. */}
{disk.reallocated_sectors !== undefined && (disk.rotation_rate ?? 0) > 0 && (
<div className="flex items-baseline justify-between gap-3">
<span className="text-[11px] uppercase tracking-wider text-muted-foreground">
Realloc. Sectors
</span>
<span
className={`font-medium flex items-center gap-1.5 ${
counterTone(disk.reallocated_sectors) === "ok"
? "text-green-500"
: counterTone(disk.reallocated_sectors) === "warn"
? "text-yellow-500"
: "text-red-500"
}`}
>
<StatusDot tone={counterTone(disk.reallocated_sectors)} />
{disk.reallocated_sectors}
</span>
</div>
)}
</div>
{/* Footer: serial (left, white mono for legibility) + observations
+ arrow-only CTA (language-neutral, compact). */}
<div className="border-t border-border/60 mt-auto pt-3 flex items-center justify-between gap-3">
{cleanSerial && cleanSerial !== "Unknown" ? (
<span className="text-[11px] text-foreground font-mono truncate min-w-0">
<span className="text-muted-foreground">S/N:</span> {cleanSerial}
</span>
) : (
<span />
)}
<div className="flex items-center gap-2 shrink-0">
{(disk.observations_count ?? 0) > 0 && (
<Badge className="bg-blue-500/10 text-blue-400 border-blue-500/20 gap-1 text-[10px]">
<Info className="h-3 w-3" />
{disk.observations_count}
</Badge>
)}
<span
className="text-blue-400 hover:text-blue-300 transition-colors text-base leading-none"
aria-label="View details"
>
</span>
</div>
</div>
</div>
)
}
const getTempColor = (temp: number, diskName?: string, rotationRate?: number) => {
if (temp === 0) return "text-gray-500"
@@ -299,12 +593,24 @@ export function StorageOverview() {
const formatHours = (hours: number) => {
if (hours === 0) return "N/A"
const years = Math.floor(hours / 8760)
const days = Math.floor((hours % 8760) / 24)
// Render in years + months when ≥1 year (e.g. "2y 6m" instead of
// "2y 189d" — months are easier to picture than triple-digit
// residual days). Months use 30.44 d/mo average to round cleanly.
// <30 days: keep days. 30 d1 yr: months + residual days when both
// values are meaningful.
const totalDays = Math.floor(hours / 24)
if (totalDays < 30) return `${totalDays}d`
const years = Math.floor(totalDays / 365)
const remainingAfterYears = totalDays - years * 365
const months = Math.floor(remainingAfterYears / 30)
if (years > 0) {
return `${years}y ${days}d`
return months > 0 ? `${years}y ${months}m` : `${years}y`
}
return `${days}d`
// Sub-year: show months + residual days if both are non-trivial.
const residualDays = remainingAfterYears - months * 30
if (months > 0 && residualDays > 0) return `${months}m ${residualDays}d`
if (months > 0) return `${months}m`
return `${totalDays}d`
}
const formatRotationRate = (rpm: number | undefined) => {
@@ -312,21 +618,12 @@ export function StorageOverview() {
return `${rpm.toLocaleString()} RPM`
}
const getDiskType = (diskName: string, rotationRate: number | undefined): string => {
if (diskName.startsWith("nvme")) {
return "NVMe"
}
// rotation_rate = -1 means HDD but RPM is unknown (detected via kernel rotational flag)
// rotation_rate = 0 or undefined means SSD
// rotation_rate > 0 means HDD with known RPM
if (rotationRate === -1) {
return "HDD"
}
if (!rotationRate || rotationRate === 0) {
return "SSD"
}
return "HDD"
}
// Thin wrapper over the shared classifier so the rest of the file
// doesn't need to be touched. The actual rules live in
// lib/disk-type.ts (single source of truth across Storage page,
// Hardware page, and any future consumer).
const getDiskType = (diskName: string, rotationRate: number | undefined): string =>
resolveDiskType(diskName, rotationRate)
const getDiskTypeBadge = (diskName: string, rotationRate: number | undefined) => {
const diskType = getDiskType(diskName, rotationRate)
@@ -690,7 +987,7 @@ export function StorageOverview() {
return (
<div className="space-y-6">
{/* Storage Summary */}
<div className="grid grid-cols-1 sm:grid-cols-2 lg:grid-cols-4 gap-3 lg:gap-6">
<div className="grid grid-cols-1 sm:grid-cols-2 xl:grid-cols-4 gap-3 xl:gap-6">
{/* ── Total Storage (preview restyle: headline + stacked bar Local·Remote·Free) ── */}
{(() => {
const totalGB = (totalLocalCapacity || 0) + (totalRemoteCapacity || 0)
@@ -1345,206 +1642,10 @@ export function StorageOverview() {
</CardTitle>
</CardHeader>
<CardContent>
<div className="space-y-4">
{storageData.disks.filter(d => d.connection_type !== 'usb').map((disk) => (
<div key={disk.name}>
<div
className="sm:hidden border border-white/10 rounded-lg p-4 cursor-pointer bg-white/5 transition-colors"
onClick={() => handleDiskClick(disk)}
>
<div className="space-y-2 mb-3">
{/* Row 1: Device name and type badge */}
<div className="flex items-center gap-2 flex-wrap">
<HardDrive className="h-5 w-5 text-muted-foreground flex-shrink-0" />
<h3 className="font-semibold">/dev/{disk.name}</h3>
<Badge className={getDiskTypeBadge(disk.name, disk.rotation_rate).className}>
{getDiskTypeBadge(disk.name, disk.rotation_rate).label}
</Badge>
{disk.is_system_disk && (
<Badge className="bg-orange-500/10 text-orange-500 border-orange-500/20 gap-1">
<Server className="h-3 w-3" />
System
</Badge>
)}
</div>
{/* Row 2: Model, temperature, and health status */}
<div className="flex items-center justify-between gap-3 pl-7">
{disk.model && disk.model !== "Unknown" && (
<p className="text-sm text-muted-foreground truncate flex-1 min-w-0">{disk.model}</p>
)}
<div className="flex items-center gap-3 flex-shrink-0">
{disk.temperature > 0 && (
<div className="flex items-center gap-1">
<Thermometer
className={`h-4 w-4 ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`}
/>
<span
className={`text-sm font-medium ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`}
>
{disk.temperature}°C
</span>
</div>
)}
{(disk.observations_count ?? 0) > 0 && (
<Badge className="bg-blue-500/10 text-blue-400 border-blue-500/20 gap-1">
<Info className="h-3 w-3" />
{disk.observations_count} obs.
</Badge>
)}
{getHealthBadge(disk.health)}
</div>
</div>
</div>
{disk.io_errors && disk.io_errors.count > 0 && (
<div className={`flex items-start gap-2 p-2 rounded text-xs ${
disk.io_errors.severity === 'CRITICAL'
? 'bg-red-500/10 text-red-400 border border-red-500/20'
: 'bg-yellow-500/10 text-yellow-400 border border-yellow-500/20'
}`}>
<AlertTriangle className="h-3.5 w-3.5 flex-shrink-0 mt-0.5" />
<span>
{disk.io_errors.error_type === 'filesystem'
? `Filesystem corruption detected`
: `${disk.io_errors.count} I/O error${disk.io_errors.count !== 1 ? 's' : ''} in 5 min`}
</span>
</div>
)}
<div className="grid grid-cols-2 gap-4 text-sm">
{disk.size_formatted && (
<div>
<p className="text-sm text-muted-foreground">Size</p>
<p className="font-medium">{disk.size_formatted}</p>
</div>
)}
{disk.smart_status && disk.smart_status !== "unknown" && (
<div>
<p className="text-sm text-muted-foreground">SMART Status</p>
<p className="font-medium capitalize">{disk.smart_status}</p>
</div>
)}
{disk.power_on_hours !== undefined && disk.power_on_hours > 0 && (
<div>
<p className="text-sm text-muted-foreground">Power On Time</p>
<p className="font-medium">{formatHours(disk.power_on_hours)}</p>
</div>
)}
{disk.serial && disk.serial !== "Unknown" && (
<div>
<p className="text-sm text-muted-foreground">Serial</p>
<p className="font-medium text-xs">{disk.serial}</p>
</div>
)}
</div>
</div>
<div
className="hidden sm:block border border-white/10 rounded-lg p-4 cursor-pointer bg-card hover:bg-white/5 transition-colors"
onClick={() => handleDiskClick(disk)}
>
<div className="space-y-2 mb-3">
{/* Row 1: Device name and type badge */}
<div className="flex items-center gap-2">
<HardDrive className="h-5 w-5 text-muted-foreground flex-shrink-0" />
<h3 className="font-semibold">/dev/{disk.name}</h3>
<Badge className={getDiskTypeBadge(disk.name, disk.rotation_rate).className}>
{getDiskTypeBadge(disk.name, disk.rotation_rate).label}
</Badge>
{disk.is_system_disk && (
<Badge className="bg-orange-500/10 text-orange-500 border-orange-500/20 gap-1">
<Server className="h-3 w-3" />
System
</Badge>
)}
</div>
{/* Row 2: Model, temperature, and health status */}
<div className="flex items-center justify-between gap-3 pl-7">
{disk.model && disk.model !== "Unknown" && (
<p className="text-sm text-muted-foreground truncate flex-1 min-w-0">{disk.model}</p>
)}
<div className="flex items-center gap-3 flex-shrink-0">
{disk.temperature > 0 && (
<div className="flex items-center gap-1">
<Thermometer
className={`h-4 w-4 ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`}
/>
<span
className={`text-sm font-medium ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`}
>
{disk.temperature}°C
</span>
</div>
)}
{(disk.observations_count ?? 0) > 0 && (
<Badge className="bg-blue-500/10 text-blue-400 border-blue-500/20 gap-1">
<Info className="h-3 w-3" />
{disk.observations_count} obs.
</Badge>
)}
{getHealthBadge(disk.health)}
</div>
</div>
</div>
{disk.io_errors && disk.io_errors.count > 0 && (
<div className={`flex items-start gap-2 p-2 rounded text-xs ${
disk.io_errors.severity === 'CRITICAL'
? 'bg-red-500/10 text-red-400 border border-red-500/20'
: 'bg-yellow-500/10 text-yellow-400 border border-yellow-500/20'
}`}>
<AlertTriangle className="h-3.5 w-3.5 flex-shrink-0 mt-0.5" />
<div>
{disk.io_errors.error_type === 'filesystem' ? (
<>
<span className="font-medium">Filesystem corruption detected</span>
{disk.io_errors.reason && (
<p className="mt-0.5 opacity-90 whitespace-pre-line">{disk.io_errors.reason}</p>
)}
</>
) : (
<>
<span className="font-medium">{disk.io_errors.count} I/O error{disk.io_errors.count !== 1 ? 's' : ''} in 5 min</span>
{disk.io_errors.sample && (
<p className="mt-0.5 opacity-80 font-mono truncate max-w-md">{disk.io_errors.sample}</p>
)}
</>
)}
</div>
</div>
)}
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 text-sm">
{disk.size_formatted && (
<div>
<p className="text-sm text-muted-foreground">Size</p>
<p className="font-medium">{disk.size_formatted}</p>
</div>
)}
{disk.smart_status && disk.smart_status !== "unknown" && (
<div>
<p className="text-sm text-muted-foreground">SMART Status</p>
<p className="font-medium capitalize">{disk.smart_status}</p>
</div>
)}
{disk.power_on_hours !== undefined && disk.power_on_hours > 0 && (
<div>
<p className="text-sm text-muted-foreground">Power On Time</p>
<p className="font-medium">{formatHours(disk.power_on_hours)}</p>
</div>
)}
{disk.serial && disk.serial !== "Unknown" && (
<div>
<p className="text-sm text-muted-foreground">Serial</p>
<p className="font-medium text-xs">{disk.serial.replace(/\\x[0-9a-fA-F]{2}/g, '')}</p>
</div>
)}
</div>
</div>
</div>
))}
<div className="grid grid-cols-1 md:grid-cols-2 lg:grid-cols-3 gap-4">
{storageData.disks
.filter((d) => d.connection_type !== 'usb')
.map((disk) => renderDiskCardV2(disk))}
</div>
</CardContent>
</Card>
@@ -1559,162 +1660,17 @@ export function StorageOverview() {
</CardTitle>
</CardHeader>
<CardContent>
<div className="space-y-4">
{storageData.disks.filter(d => d.connection_type === 'usb').map((disk) => (
<div key={disk.name}>
{/* Mobile card */}
<div
className="sm:hidden border border-white/10 rounded-lg p-4 cursor-pointer bg-white/5 transition-colors"
onClick={() => handleDiskClick(disk)}
>
<div className="space-y-2 mb-3">
<div className="flex items-center gap-2">
<Usb className="h-5 w-5 text-orange-400 flex-shrink-0" />
<h3 className="font-semibold">/dev/{disk.name}</h3>
<Badge className="bg-orange-500/10 text-orange-400 border-orange-500/20 text-[10px] px-1.5">USB</Badge>
</div>
<div className="flex items-center justify-between gap-3 pl-7">
{disk.model && disk.model !== "Unknown" && (
<p className="text-sm text-muted-foreground truncate flex-1 min-w-0">{disk.model}</p>
)}
<div className="flex items-center gap-3 flex-shrink-0">
{disk.temperature > 0 && (
<div className="flex items-center gap-1">
<Thermometer className={`h-4 w-4 ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`} />
<span className={`text-sm font-medium ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`}>
{disk.temperature}°C
</span>
</div>
)}
{(disk.observations_count ?? 0) > 0 && (
<Badge className="bg-blue-500/10 text-blue-400 border-blue-500/20 gap-1">
<Info className="h-3 w-3" />
{disk.observations_count}
</Badge>
)}
{getHealthBadge(disk.health)}
</div>
</div>
</div>
{/* USB Mobile: Size, SMART, Serial grid */}
<div className="grid grid-cols-2 gap-4 text-sm">
{disk.size_formatted && (
<div>
<p className="text-sm text-muted-foreground">Size</p>
<p className="font-medium">{disk.size_formatted}</p>
</div>
)}
{disk.smart_status && disk.smart_status !== "unknown" && (
<div>
<p className="text-sm text-muted-foreground">SMART Status</p>
<p className="font-medium capitalize">{disk.smart_status}</p>
</div>
)}
{disk.serial && disk.serial !== "Unknown" && (
<div>
<p className="text-sm text-muted-foreground">Serial</p>
<p className="font-medium text-xs">{disk.serial.replace(/\\x[0-9a-fA-F]{2}/g, '')}</p>
</div>
)}
</div>
</div>
{/* Desktop */}
<div
className="hidden sm:block border border-white/10 rounded-lg p-4 cursor-pointer hover:bg-white/5 transition-colors"
onClick={() => handleDiskClick(disk)}
>
<div className="flex items-center justify-between mb-3">
<div className="flex items-center gap-2">
<Usb className="h-5 w-5 text-orange-400" />
<h3 className="font-semibold">/dev/{disk.name}</h3>
<Badge className="bg-orange-500/10 text-orange-400 border-orange-500/20 text-[10px] px-1.5">USB</Badge>
</div>
<div className="flex items-center gap-3">
{disk.temperature > 0 && (
<div className="flex items-center gap-1">
<Thermometer className={`h-4 w-4 ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`} />
<span className={`text-sm font-medium ${getTempColor(disk.temperature, disk.name, disk.rotation_rate)}`}>
{disk.temperature}°C
</span>
</div>
)}
{getHealthBadge(disk.health)}
{(disk.observations_count ?? 0) > 0 && (
<Badge className="bg-blue-500/10 text-blue-400 border-blue-500/20 gap-1">
<Info className="h-3 w-3" />
{disk.observations_count} obs.
</Badge>
)}
</div>
</div>
{disk.model && disk.model !== "Unknown" && (
<p className="text-sm text-muted-foreground mb-3 ml-7">{disk.model}</p>
)}
{disk.io_errors && disk.io_errors.count > 0 && (
<div className={`flex items-start gap-2 p-2 rounded text-xs mb-3 ${
disk.io_errors.severity === 'CRITICAL'
? 'bg-red-500/10 text-red-400 border border-red-500/20'
: 'bg-yellow-500/10 text-yellow-400 border border-yellow-500/20'
}`}>
<AlertTriangle className="h-3.5 w-3.5 flex-shrink-0 mt-0.5" />
<div>
{disk.io_errors.error_type === 'filesystem' ? (
<>
<span className="font-medium">Filesystem corruption detected</span>
{disk.io_errors.reason && (
<p className="mt-0.5 opacity-90 whitespace-pre-line">{disk.io_errors.reason}</p>
)}
</>
) : (
<>
<span className="font-medium">{disk.io_errors.count} I/O error{disk.io_errors.count !== 1 ? 's' : ''} in 5 min</span>
{disk.io_errors.sample && (
<p className="mt-0.5 opacity-80 font-mono truncate max-w-md">{disk.io_errors.sample}</p>
)}
</>
)}
</div>
</div>
)}
<div className="grid grid-cols-2 md:grid-cols-4 gap-4 text-sm">
{disk.size_formatted && (
<div>
<p className="text-sm text-muted-foreground">Size</p>
<p className="font-medium">{disk.size_formatted}</p>
</div>
)}
{disk.smart_status && disk.smart_status !== "unknown" && (
<div>
<p className="text-sm text-muted-foreground">SMART Status</p>
<p className="font-medium capitalize">{disk.smart_status}</p>
</div>
)}
{disk.power_on_hours !== undefined && disk.power_on_hours > 0 && (
<div>
<p className="text-sm text-muted-foreground">Power On Time</p>
<p className="font-medium">{formatHours(disk.power_on_hours)}</p>
</div>
)}
{disk.serial && disk.serial !== "Unknown" && (
<div>
<p className="text-sm text-muted-foreground">Serial</p>
<p className="font-medium text-xs">{disk.serial.replace(/\\x[0-9a-fA-F]{2}/g, '')}</p>
</div>
)}
</div>
</div>
</div>
))}
<div className="grid grid-cols-1 md:grid-cols-2 lg:grid-cols-3 gap-4">
{storageData.disks
.filter((d) => d.connection_type === 'usb')
.map((disk) => renderDiskCardV2(disk))}
</div>
</CardContent>
</Card>
)}
{/* Disk Details Dialog */}
{/* Disk Details Dialog wider on desktop so the SMART chart +
tabs have room without the right column hugging the edge. */}
<Dialog open={detailsOpen} onOpenChange={(open) => {
setDetailsOpen(open)
if (!open) {
@@ -1722,7 +1678,7 @@ export function StorageOverview() {
setSmartJsonData(null)
}
}}>
<DialogContent className="max-w-2xl max-h-[80vh] sm:max-h-[85vh] overflow-hidden flex flex-col p-0">
<DialogContent className="max-w-4xl max-h-[80vh] sm:max-h-[85vh] overflow-hidden flex flex-col p-0">
<DialogHeader className="px-6 pt-6 pb-0">
<DialogTitle className="flex items-center gap-2">
{selectedDisk?.connection_type === 'usb' ? (
@@ -2047,29 +2003,53 @@ export function StorageOverview() {
</div>
<div>
<p className="text-sm text-muted-foreground">SMART Status</p>
<p className="font-medium capitalize">{selectedDisk.smart_status}</p>
<p className={`font-medium capitalize flex items-center gap-1.5 ${
smartStatusTone(selectedDisk.smart_status) === "ok"
? "text-green-500"
: smartStatusTone(selectedDisk.smart_status) === "fail"
? "text-red-500"
: ""
}`}>
<StatusDot tone={smartStatusTone(selectedDisk.smart_status)} />
{selectedDisk.smart_status}
</p>
</div>
<div>
<p className="text-sm text-muted-foreground">Reallocated Sectors</p>
<p
className={`font-medium ${selectedDisk.reallocated_sectors && selectedDisk.reallocated_sectors > 0 ? "text-yellow-500" : ""}`}
>
<p className={`font-medium flex items-center gap-1.5 ${
counterTone(selectedDisk.reallocated_sectors) === "ok"
? "text-green-500"
: counterTone(selectedDisk.reallocated_sectors) === "warn"
? "text-yellow-500"
: "text-red-500"
}`}>
<StatusDot tone={counterTone(selectedDisk.reallocated_sectors)} />
{selectedDisk.reallocated_sectors ?? 0}
</p>
</div>
<div>
<p className="text-sm text-muted-foreground">Pending Sectors</p>
<p
className={`font-medium ${selectedDisk.pending_sectors && selectedDisk.pending_sectors > 0 ? "text-yellow-500" : ""}`}
>
<p className={`font-medium flex items-center gap-1.5 ${
counterTone(selectedDisk.pending_sectors) === "ok"
? "text-green-500"
: counterTone(selectedDisk.pending_sectors) === "warn"
? "text-yellow-500"
: "text-red-500"
}`}>
<StatusDot tone={counterTone(selectedDisk.pending_sectors)} />
{selectedDisk.pending_sectors ?? 0}
</p>
</div>
<div>
<p className="text-sm text-muted-foreground">CRC Errors</p>
<p
className={`font-medium ${selectedDisk.crc_errors && selectedDisk.crc_errors > 0 ? "text-yellow-500" : ""}`}
>
<p className={`font-medium flex items-center gap-1.5 ${
counterTone(selectedDisk.crc_errors) === "ok"
? "text-green-500"
: counterTone(selectedDisk.crc_errors) === "warn"
? "text-yellow-500"
: "text-red-500"
}`}>
<StatusDot tone={counterTone(selectedDisk.crc_errors)} />
{selectedDisk.crc_errors ?? 0}
</p>
</div>
@@ -3737,7 +3717,7 @@ ${observationsHtml}
<!-- Footer -->
<div class="rpt-footer">
<div>Report generated by ProxMenux Monitor</div>
<div>ProxMenux Monitor v1.2.2.1-beta</div>
<div>ProxMenux Monitor v1.2.4</div>
</div>
</body>
+1 -1
View File
@@ -591,7 +591,7 @@ export function SystemLogs() {
)}
{/* Statistics Cards */}
<div className="grid grid-cols-2 lg:grid-cols-4 gap-4 lg:gap-6">
<div className="grid grid-cols-2 xl:grid-cols-4 gap-4 xl:gap-6">
<Card className="bg-card border-border">
<CardHeader className="flex flex-row items-center justify-between space-y-0 pb-2">
<CardTitle className="text-sm font-medium text-muted-foreground">Total Entries</CardTitle>
+43 -9
View File
@@ -4,10 +4,11 @@ import { useState, useEffect } from "react"
import { Card, CardContent, CardHeader, CardTitle } from "./ui/card"
import { Progress } from "./ui/progress"
import { Badge } from "./ui/badge"
import { Cpu, MemoryStick, Thermometer, Server, Zap, AlertCircle, HardDrive, Network } from "lucide-react"
import { Cpu, MemoryStick, Thermometer, Server, Zap, AlertCircle, HardDrive, Network, ChevronRight } from "lucide-react"
import { NodeMetricsCharts } from "./node-metrics-charts"
import { NetworkTrafficChart } from "./network-traffic-chart"
import { TemperatureDetailModal } from "./temperature-detail-modal"
import { ProcessDetailModal } from "./process-detail-modal"
import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "./ui/select"
import { fetchApi } from "../lib/api-config"
import { formatNetworkTraffic, getNetworkUnit } from "../lib/format-network"
@@ -187,6 +188,8 @@ export function SystemOverview() {
const [networkTotals, setNetworkTotals] = useState<{ received: number; sent: number }>({ received: 0, sent: 0 })
const [networkUnit, setNetworkUnit] = useState<"Bytes" | "Bits">("Bytes") // Added networkUnit state
const [tempModalOpen, setTempModalOpen] = useState(false)
const [cpuProcModalOpen, setCpuProcModalOpen] = useState(false)
const [memProcModalOpen, setMemProcModalOpen] = useState(false)
useEffect(() => {
const fetchAllData = async () => {
@@ -398,12 +401,19 @@ export function SystemOverview() {
return (
<div className="space-y-6">
<div className="grid grid-cols-1 sm:grid-cols-2 lg:grid-cols-4 gap-3 lg:gap-6">
<div className="grid grid-cols-1 sm:grid-cols-2 xl:grid-cols-4 gap-3 xl:gap-6">
{/* ── CPU Usage (preview restyle v2: tamaño igual a System Info, bars más anchas) ── */}
<Card className="bg-card border-border">
<Card
className="bg-card border-border cursor-pointer hover:bg-white/5 transition-colors"
onClick={() => setCpuProcModalOpen(true)}
title="View top processes by CPU"
>
<CardHeader className="flex flex-row items-center justify-between space-y-0 pb-2">
<CardTitle className="text-sm font-medium text-muted-foreground">CPU Usage</CardTitle>
<Cpu className="h-4 w-4 text-muted-foreground" />
<div className="flex items-center gap-1 text-muted-foreground">
<Cpu className="h-4 w-4" />
<ChevronRight className="h-4 w-4 opacity-60" />
</div>
</CardHeader>
<CardContent>
<div className="flex items-center gap-4">
@@ -443,10 +453,17 @@ export function SystemOverview() {
</Card>
{/* ── Memory (preview restyle v2: tamaño igual a System Info, bars más anchas) ── */}
<Card className="bg-card border-border">
<Card
className="bg-card border-border cursor-pointer hover:bg-white/5 transition-colors"
onClick={() => setMemProcModalOpen(true)}
title="View top processes by memory"
>
<CardHeader className="flex flex-row items-center justify-between space-y-0 pb-2">
<CardTitle className="text-sm font-medium text-muted-foreground">Memory</CardTitle>
<MemoryStick className="h-4 w-4 text-muted-foreground" />
<div className="flex items-center gap-1 text-muted-foreground">
<MemoryStick className="h-4 w-4" />
<ChevronRight className="h-4 w-4 opacity-60" />
</div>
</CardHeader>
<CardContent>
<div className="flex items-center gap-4">
@@ -524,7 +541,12 @@ export function SystemOverview() {
>
<CardHeader className="flex flex-row items-center justify-between space-y-0 pb-2">
<CardTitle className="text-sm font-medium text-muted-foreground">Temperature</CardTitle>
<Thermometer className="h-4 w-4 text-muted-foreground" />
<div className="flex items-center gap-1 text-muted-foreground">
<Thermometer className="h-4 w-4" />
{systemData.temperature > 0 && (
<ChevronRight className="h-4 w-4 opacity-60" />
)}
</div>
</CardHeader>
<CardContent>
<div className="flex items-center justify-between">
@@ -566,12 +588,24 @@ export function SystemOverview() {
</Card>
</div>
<TemperatureDetailModal
open={tempModalOpen}
<TemperatureDetailModal
open={tempModalOpen}
onOpenChange={setTempModalOpen}
liveTemperature={systemData.temperature}
/>
<ProcessDetailModal
open={cpuProcModalOpen}
onOpenChange={setCpuProcModalOpen}
sort="cpu"
/>
<ProcessDetailModal
open={memProcModalOpen}
onOpenChange={setMemProcModalOpen}
sort="mem"
/>
<NodeMetricsCharts />
<div className="grid grid-cols-1 lg:grid-cols-2 gap-6">
@@ -70,7 +70,10 @@ const getStatusInfo = (temp: number) => {
}
export function TemperatureDetailModal({ open, onOpenChange, liveTemperature }: TemperatureDetailModalProps) {
const [timeframe, setTimeframe] = useState("hour")
// Default to 24 h — matches the disk temperature modal and is the
// useful timeframe for spotting trends; the 1-h view rarely tells
// you anything that the live reading doesn't already show.
const [timeframe, setTimeframe] = useState("day")
const [data, setData] = useState<TempHistoryPoint[]>([])
const [stats, setStats] = useState<TempStats>({ min: 0, max: 0, avg: 0, current: 0 })
const [loading, setLoading] = useState(true)
+39 -13
View File
@@ -92,33 +92,59 @@ export function TwoFactorSetup({ open, onClose, onSuccess }: TwoFactorSetupProps
const copyToClipboard = async (text: string, type: "secret" | "codes") => {
let ok = false
// Preferred path (HTTPS / localhost). On plain HTTP the Promise rejects,
// so we catch and fall through to the textarea fallback.
// Path 1: modern Clipboard API. Only works on HTTPS / localhost.
try {
if (navigator.clipboard && window.isSecureContext) {
if (navigator.clipboard?.writeText) {
await navigator.clipboard.writeText(text)
ok = true
}
} catch {
// fall through to execCommand fallback
// fall through
}
// Path 2: legacy execCommand. Picky — some browsers (iOS Safari
// especially) refuse to copy from an element placed off-screen
// (`left: -9999px`), which is the previous version's mistake.
// Keep the textarea inside the viewport but visually invisible.
if (!ok) {
const textarea = document.createElement("textarea")
textarea.value = text
textarea.style.position = "fixed"
textarea.style.top = "0"
textarea.style.left = "0"
textarea.style.width = "2em"
textarea.style.height = "2em"
textarea.style.padding = "0"
textarea.style.border = "none"
textarea.style.outline = "none"
textarea.style.boxShadow = "none"
textarea.style.background = "transparent"
textarea.style.opacity = "0"
textarea.setAttribute("readonly", "")
textarea.setAttribute("aria-hidden", "true")
document.body.appendChild(textarea)
try {
const textarea = document.createElement("textarea")
textarea.value = text
textarea.style.position = "fixed"
textarea.style.left = "-9999px"
textarea.style.top = "-9999px"
textarea.style.opacity = "0"
textarea.readOnly = true
document.body.appendChild(textarea)
textarea.focus()
textarea.select()
textarea.setSelectionRange(0, text.length)
ok = document.execCommand("copy")
document.body.removeChild(textarea)
} catch {
ok = false
} finally {
document.body.removeChild(textarea)
}
}
// Path 3: last-resort window.prompt — ugly but unblockable. The
// user can select+copy from the prompt manually. This guarantees
// they can finish the 2FA setup even on plain-HTTP Monitor where
// both the Clipboard API and execCommand may be locked down.
if (!ok) {
try {
window.prompt("Copy this value:", text)
ok = true
} catch {
// ignore
}
}
+8 -1
View File
@@ -9,7 +9,14 @@ const Input = React.forwardRef<HTMLInputElement, InputProps>(({ className, type,
<input
type={type}
className={cn(
"flex h-10 w-full rounded-lg border border-input bg-background px-4 py-2 text-sm shadow-sm transition-all file:border-0 file:bg-transparent file:text-sm file:font-medium placeholder:text-muted-foreground focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 disabled:cursor-not-allowed disabled:opacity-50 hover:border-ring/50",
// The previous focus style was `ring-2 ring-ring ring-offset-2`, which
// painted a 2px white ring with a 2px gap outside the border. Inside a
// ScrollArea or any container with `overflow-hidden` the ring's left
// edge got clipped and the result looked broken. We replace it with a
// 1px blue ring + matching border so a focused input now sits at the
// same visual weight as the colored card selectors used elsewhere
// (Backend picker, etc.).
"flex h-10 w-full rounded-lg border border-input bg-background px-4 py-2 text-sm shadow-sm transition-all file:border-0 file:bg-transparent file:text-sm file:font-medium placeholder:text-muted-foreground focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-blue-500 focus-visible:border-blue-500 disabled:cursor-not-allowed disabled:opacity-50 hover:border-ring/50",
className,
)}
ref={ref}
+1 -1
View File
@@ -11,7 +11,7 @@ const Switch = React.forwardRef<
>(({ className, ...props }, ref) => (
<SwitchPrimitives.Root
className={cn(
"peer inline-flex h-5 w-9 shrink-0 cursor-pointer items-center rounded-full border-2 border-transparent shadow-sm transition-colors focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 focus-visible:ring-offset-background disabled:cursor-not-allowed disabled:opacity-50 data-[state=checked]:bg-primary data-[state=unchecked]:bg-input",
"peer inline-flex h-5 w-9 shrink-0 cursor-pointer items-center rounded-full border-2 border-transparent shadow-sm transition-colors focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 focus-visible:ring-offset-background disabled:cursor-not-allowed disabled:opacity-50 data-[state=checked]:bg-primary data-[state=unchecked]:bg-slate-300 dark:data-[state=unchecked]:bg-slate-600",
className
)}
{...props}
+1 -1
View File
@@ -10,7 +10,7 @@ const Textarea = React.forwardRef<HTMLTextAreaElement, TextareaProps>(
return (
<textarea
className={cn(
"flex min-h-[80px] w-full rounded-md border border-input bg-background px-3 py-2 text-sm ring-offset-background placeholder:text-muted-foreground focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2 disabled:cursor-not-allowed disabled:opacity-50",
"flex min-h-[80px] w-full rounded-md border border-input bg-background px-3 py-2 text-sm placeholder:text-muted-foreground focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-blue-500 focus-visible:border-blue-500 disabled:cursor-not-allowed disabled:opacity-50",
className
)}
ref={ref}
+224 -79
View File
@@ -8,7 +8,7 @@ import { Badge } from "./ui/badge"
import { Progress } from "./ui/progress"
import { Button } from "./ui/button"
import { Dialog, DialogContent, DialogHeader, DialogTitle, DialogFooter, DialogDescription } from "./ui/dialog"
import { Server, Play, Square, Cpu, MemoryStick, HardDrive, Network, Power, RotateCcw, StopCircle, Container, ChevronDown, ChevronUp, ChevronRight, Terminal, Archive, Plus, Loader2, Clock, Database, Shield, Bell, FileText, Settings2, Activity, Package, RefreshCw } from 'lucide-react'
import { Server, Play, Square, Cpu, MemoryStick, HardDrive, Network, Power, RotateCcw, StopCircle, Container, ChevronDown, ChevronUp, ChevronRight, Terminal, Archive, Plus, Loader2, Clock, Database, Shield, Bell, FileText, Settings2, Activity, Package, RefreshCw, EthernetPort } from 'lucide-react'
import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "./ui/select"
import { Checkbox } from "./ui/checkbox"
import { Textarea } from "./ui/textarea"
@@ -1129,7 +1129,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
.reduce((sum, vm) => sum + (vm.maxmem || 0), 0) / 1024 ** 3).toFixed(1)
}, [safeVMData])
const { data: systemData } = useSWR<{ memory_total: number; memory_used: number; memory_usage: number }>(
const { data: systemData } = useSWR<{ memory_total: number; memory_used: number; memory_usage: number; cpu_cores?: number; cpu_threads?: number }>(
"/api/system",
fetcher,
{
@@ -1301,7 +1301,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
}
`}</style>
<div className="grid grid-cols-1 sm:grid-cols-2 lg:grid-cols-4 gap-6">
<div className="grid grid-cols-1 sm:grid-cols-2 xl:grid-cols-4 gap-6">
{/* ── Total VMs & LXCs (preview restyle: B-headline + pills, matching Overview) ── */}
{(() => {
const running = safeVMData.filter((vm) => vm.status === "running").length
@@ -1346,6 +1346,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
const inUseVCPU = safeVMData
.filter((vm) => vm.status === "running")
.reduce((sum, vm) => sum + (vm.maxcpu || 0), 0)
const hostThreads = systemData?.cpu_threads ?? systemData?.cpu_cores ?? 0
const stroke = allocPct >= 90 ? '#ef4444' : allocPct >= 75 ? '#f59e0b' : '#3b82f6'
return (
<Card className="bg-card border-border">
@@ -1374,11 +1375,11 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
</div>
<div className="flex items-center justify-between text-sm">
<span className="text-muted-foreground">Configured</span>
<span className="font-medium font-mono whitespace-nowrap">{configuredVCPU || '—'} vCPU</span>
<span className="font-medium font-mono whitespace-nowrap">{configuredVCPU || '—'}{hostThreads ? ` / ${hostThreads}` : ''} vCPU</span>
</div>
<div className="flex items-center justify-between text-sm">
<span className="text-muted-foreground">In use</span>
<span className="font-medium font-mono whitespace-nowrap">{inUseVCPU || '—'} vCPU</span>
<span className="font-medium font-mono whitespace-nowrap">{inUseVCPU || '—'}{hostThreads ? ` / ${hostThreads}` : ''} vCPU</span>
</div>
</div>
</div>
@@ -1979,11 +1980,13 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
/>
</div>
{/* Disk I/O */}
{/* Disk I/O cumulative counters from Proxmox
API: bytes read/written since the VM/LXC
was last started, not a per-period rate. */}
<div>
<div className="flex items-center gap-1.5 text-xs text-muted-foreground mb-2">
<HardDrive className="h-3.5 w-3.5" />
<span>Disk I/O</span>
<span>Disk I/O <span className="text-[10px] opacity-70">(since boot)</span></span>
</div>
<div className="space-y-1">
<div className="text-sm text-green-500 flex items-center gap-1">
@@ -1997,11 +2000,13 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
</div>
</div>
{/* Network I/O */}
{/* Network I/O cumulative counters from
Proxmox API: bytes in/out since the VM/LXC
was last started. */}
<div>
<div className="flex items-center gap-1.5 text-xs text-muted-foreground mb-2">
<Network className="h-3.5 w-3.5" />
<span>Network I/O</span>
<span>Network I/O <span className="text-[10px] opacity-70">(since boot)</span></span>
</div>
<div className="space-y-1">
<div className="text-sm text-green-500 flex items-center gap-1">
@@ -2468,71 +2473,144 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
</div>
</div>
{/* Storage Section */}
<div>
<h4 className="flex items-center gap-2 text-sm font-semibold text-muted-foreground mb-3 uppercase tracking-wide">
<HardDrive className="h-4 w-4" />
Storage
</h4>
<div className="space-y-3">
{vmDetails.config.rootfs && (
<div key="rootfs">
<div className="text-xs text-muted-foreground mb-1">Root Filesystem</div>
<div className="font-medium text-foreground text-sm break-all font-mono bg-muted/50 p-2 rounded">
{vmDetails.config.rootfs}
{/* Storage Section human-readable breakdown
per disk plus the raw config string in a
collapsible details block, mirroring the
Network section. */}
{(() => {
// Parse a Proxmox disk config string into
// { storage, volume, path, options }
// Handles both LVM-style volumes
// "local-lvm:vm-101-disk-0,size=6G" and
// passthrough paths "/dev/disk/by-id/...".
const parseDisk = (raw: string) => {
const parts = raw.split(",")
const first = parts[0] || ""
const options: Record<string, string> = {}
parts.slice(1).forEach((p) => {
const eq = p.indexOf("=")
if (eq > 0) {
options[p.slice(0, eq).trim()] = p.slice(eq + 1).trim()
}
})
let storage = "", volume = "", path = ""
if (first.startsWith("/")) {
path = first
} else if (first.includes(":")) {
const [s, v] = first.split(":")
storage = s
volume = v
} else {
volume = first
}
return { storage, volume, path, options }
}
// Convert Proxmox size strings ("6G",
// "3907018584K", "40G", "4M") to a
// consistent GB/TB display.
const humanSize = (s: string): string => {
if (!s) return ""
const m = s.match(/^(\d+(?:\.\d+)?)([KMGT])?$/i)
if (!m) return s
const n = parseFloat(m[1])
const unit = (m[2] || "").toUpperCase()
const bytes =
unit === "K" ? n * 1024 :
unit === "M" ? n * 1024 ** 2 :
unit === "G" ? n * 1024 ** 3 :
unit === "T" ? n * 1024 ** 4 : n
if (bytes >= 1024 ** 4) return `${(bytes / 1024 ** 4).toFixed(2)} TB`
if (bytes >= 1024 ** 3) return `${(bytes / 1024 ** 3).toFixed(bytes < 10 * 1024 ** 3 ? 1 : 0)} GB`
if (bytes >= 1024 ** 2) return `${(bytes / 1024 ** 2).toFixed(0)} MB`
return s
}
const DField = ({ label, value, mono, className }:
{ label: string; value: string; mono?: boolean; className?: string }) => (
<div className="flex flex-col gap-0.5">
<span className="text-[10px] uppercase tracking-wide text-muted-foreground">{label}</span>
<span className={`text-foreground ${mono ? "font-mono text-xs" : "text-sm"} ${className || ""}`}>{value}</span>
</div>
)
const renderDisk = (label: string, raw: string, keyId: string) => {
const d = parseDisk(raw)
return (
<div key={keyId} className="bg-muted/30 rounded-md p-3 space-y-3">
<div className="flex items-center gap-2 flex-wrap">
<HardDrive className="h-4 w-4 text-purple-500 flex-shrink-0" />
<span className="text-sm font-semibold text-foreground">{label}</span>
{d.storage && (
<span className="text-xs text-orange-500 font-mono">
{d.storage}
</span>
)}
{d.options.size && (
<span className="text-xs text-cyan-500 font-mono">
{humanSize(d.options.size)}
</span>
)}
</div>
</div>
)}
{vmDetails.config.scsihw && (
<div key="scsihw">
<div className="text-xs text-muted-foreground mb-1">SCSI Controller</div>
<div className="font-medium text-foreground">{vmDetails.config.scsihw}</div>
</div>
)}
{/* Disk Storage with proper keys */}
{Object.keys(vmDetails.config)
.filter((key) => key.match(/^(scsi|sata|ide|virtio)\d+$/))
.map((diskKey) => (
<div key={`disk-${selectedVM.vmid}-${diskKey}`}>
<div className="text-xs text-muted-foreground mb-1">
{diskKey.toUpperCase().replace(/(\d+)/, " $1")}
<div className="grid grid-cols-2 sm:grid-cols-3 gap-x-4 gap-y-2">
{d.volume && <DField label="Volume" value={d.volume} mono />}
{d.path && <DField label="Path" value={d.path} mono className="break-all" />}
{d.options.ssd === "1" && <DField label="Media" value="SSD" />}
{d.options.discard && <DField label="Discard" value={d.options.discard} />}
{d.options.iothread === "1" && <DField label="IOThread" value="on" />}
{d.options.cache && <DField label="Cache" value={d.options.cache} />}
{d.options.aio && <DField label="AIO" value={d.options.aio} />}
{d.options.backup === "0" && <DField label="Backup" value="excluded" className="text-red-500" />}
{d.options.backup === "1" && <DField label="Backup" value="included" />}
{d.options.replicate === "0" && <DField label="Replicate" value="off" />}
{d.options.efitype && <DField label="EFI type" value={d.options.efitype} />}
{d.options.pre_enrolled_keys && <DField label="Pre-enrolled keys" value={d.options.pre_enrolled_keys} />}
{d.options.serial && <DField label="Serial" value={d.options.serial} mono />}
{d.options.mp && <DField label="Mount point" value={d.options.mp} mono />}
{d.options.acl && <DField label="ACL" value={d.options.acl} />}
</div>
<details className="text-xs">
<summary className="cursor-pointer text-muted-foreground hover:text-foreground">Raw config</summary>
<div className="mt-1 font-mono text-foreground break-all bg-background/50 p-2 rounded">
{raw}
</div>
<div className="font-medium text-foreground text-sm break-all font-mono bg-muted/50 p-2 rounded">
{vmDetails.config[diskKey]}
</div>
</div>
))}
{vmDetails.config.efidisk0 && (
<div key="efidisk0">
<div className="text-xs text-muted-foreground mb-1">EFI Disk</div>
<div className="font-medium text-foreground text-sm break-all font-mono bg-muted/50 p-2 rounded">
{vmDetails.config.efidisk0}
</div>
</details>
</div>
)}
{vmDetails.config.tpmstate0 && (
<div key="tpmstate0">
<div className="text-xs text-muted-foreground mb-1">TPM State</div>
<div className="font-medium text-foreground text-sm break-all font-mono bg-muted/50 p-2 rounded">
{vmDetails.config.tpmstate0}
</div>
)
}
return (
<div>
<h4 className="flex items-center gap-2 text-sm font-semibold text-muted-foreground mb-3 uppercase tracking-wide">
<HardDrive className="h-4 w-4" />
Storage
</h4>
<div className="space-y-3">
{vmDetails.config.scsihw && (
<div className="flex items-center gap-2 text-sm">
<span className="text-xs uppercase tracking-wide text-muted-foreground">SCSI controller:</span>
<span className="font-mono text-foreground">{vmDetails.config.scsihw}</span>
</div>
)}
{vmDetails.config.rootfs && renderDisk("Root Filesystem", vmDetails.config.rootfs as string, "rootfs")}
{Object.keys(vmDetails.config)
.filter((key) => key.match(/^(scsi|sata|ide|virtio)\d+$/))
.sort()
.map((diskKey) => renderDisk(
diskKey.toUpperCase().replace(/(\d+)/, " $1"),
vmDetails.config[diskKey] as string,
`disk-${selectedVM.vmid}-${diskKey}`,
))}
{vmDetails.config.efidisk0 && renderDisk("EFI Disk", vmDetails.config.efidisk0 as string, "efidisk0")}
{vmDetails.config.tpmstate0 && renderDisk("TPM State", vmDetails.config.tpmstate0 as string, "tpmstate0")}
{Object.keys(vmDetails.config)
.filter((key) => key.match(/^mp\d+$/))
.sort()
.map((mpKey) => renderDisk(
`Mount Point ${mpKey.replace("mp", "")}`,
vmDetails.config[mpKey] as string,
`mp-${selectedVM.vmid}-${mpKey}`,
))}
</div>
)}
{/* Mount Points with proper keys */}
{Object.keys(vmDetails.config)
.filter((key) => key.match(/^mp\d+$/))
.map((mpKey) => (
<div key={`mp-${selectedVM.vmid}-${mpKey}`}>
<div className="text-xs text-muted-foreground mb-1">
Mount Point {mpKey.replace("mp", "")}
</div>
<div className="font-medium text-foreground text-sm break-all font-mono bg-muted/50 p-2 rounded">
{vmDetails.config[mpKey]}
</div>
</div>
))}
</div>
</div>
</div>
)
})()}
{/* Network Section */}
<div>
@@ -2541,19 +2619,86 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => {
Network
</h4>
<div className="space-y-3">
{/* Network Interfaces with proper keys */}
{/* Network Interfaces with proper keys.
Renders BOTH a human-readable
breakdown (bridge, IP, gw, MAC,
host iface) AND the raw config
string so power users still see
the underlying Proxmox config. */}
{Object.keys(vmDetails.config)
.filter((key) => key.match(/^net\d+$/))
.map((netKey) => (
<div key={`net-${selectedVM.vmid}-${netKey}`}>
<div className="text-xs text-muted-foreground mb-1">
Network Interface {netKey.replace("net", "")}
.map((netKey) => {
const raw = vmDetails.config[netKey] as string
// Parse "name=eth0,bridge=vmbr0,gw=1.2.3.4,..."
const parsed: Record<string, string> = {}
raw.split(",").forEach((pair) => {
const eq = pair.indexOf("=")
if (eq > 0) {
parsed[pair.slice(0, eq).trim()] =
pair.slice(eq + 1).trim()
} else if (pair && !parsed.model) {
// bare "virtio" / "e1000" → NIC model
parsed.model = pair.trim()
}
})
const idx = netKey.replace("net", "")
// For VMs, the host-side iface is
// tap<vmid>i<idx>; for LXC it's
// veth<vmid>i<idx>. Surface it so
// users can correlate with the
// "VM & LXC Network Interfaces"
// card on the network page.
const hostIface = selectedVM.type === "lxc"
? `veth${selectedVM.vmid}i${idx}`
: `tap${selectedVM.vmid}i${idx}`
const Field = ({ label, value, mono }:
{ label: string; value: string; mono?: boolean }) => (
<div className="flex flex-col gap-0.5">
<span className="text-[10px] uppercase tracking-wide text-muted-foreground">{label}</span>
<span className={`text-foreground ${mono ? "font-mono text-xs" : "text-sm"}`}>{value}</span>
</div>
<div className="font-medium text-green-500 text-sm break-all font-mono bg-muted/50 p-2 rounded">
{vmDetails.config[netKey]}
)
return (
<div key={`net-${selectedVM.vmid}-${netKey}`}
className="bg-muted/30 rounded-md p-3 space-y-3">
<div className="flex items-center gap-2 flex-wrap">
<EthernetPort className="h-4 w-4 text-green-500 flex-shrink-0" />
<span className="text-sm font-semibold text-foreground">
Network Interface {idx}
</span>
{parsed.name && (
<span className="text-xs text-muted-foreground font-mono">
({parsed.name})
</span>
)}
<span className="text-xs text-orange-500 font-mono">
host: {hostIface}
</span>
</div>
<div className="grid grid-cols-2 sm:grid-cols-3 gap-x-4 gap-y-2">
{parsed.bridge && <Field label="Bridge" value={parsed.bridge} mono />}
{parsed.ip && <Field label="IP" value={parsed.ip} mono />}
{parsed.ip6 && <Field label="IPv6" value={parsed.ip6} mono />}
{parsed.gw && <Field label="Gateway" value={parsed.gw} mono />}
{parsed.gw6 && <Field label="Gateway v6" value={parsed.gw6} mono />}
{parsed.hwaddr && <Field label="MAC" value={parsed.hwaddr.toUpperCase()} mono />}
{parsed.virtio && <Field label="MAC" value={parsed.virtio.toUpperCase()} mono />}
{parsed.e1000 && <Field label="MAC" value={parsed.e1000.toUpperCase()} mono />}
{parsed.type && <Field label="Type" value={parsed.type} mono />}
{parsed.tag && <Field label="VLAN" value={parsed.tag} mono />}
{parsed.mtu && <Field label="MTU" value={parsed.mtu} mono />}
{parsed.rate && <Field label="Rate limit" value={`${parsed.rate} MB/s`} mono />}
{parsed.firewall === "1" && <Field label="Firewall" value="enabled" />}
</div>
<details className="text-xs">
<summary className="cursor-pointer text-muted-foreground hover:text-foreground">Raw config</summary>
<div className="mt-1 font-mono text-green-500 break-all bg-background/50 p-2 rounded">
{raw}
</div>
</details>
</div>
</div>
))}
)
})}
<div className="grid grid-cols-1 lg:grid-cols-2 gap-3">
{vmDetails.config.nameserver && (
<div>
+27 -23
View File
@@ -1,30 +1,31 @@
{
"_description": "Verified AI models for ProxMenux notifications. Only models listed here will be shown to users. Models are tested to work with the chat/completions API format.",
"_updated": "2026-04-19",
"_updated": "2026-07-14",
"_verifier": "Refreshed with tools/ai-models-verifier (private). Re-run before each ProxMenux release to keep the list current. The verifier and ProxMenux share the same reasoning/thinking-model handlers so their verdicts stay aligned with runtime behaviour.",
"groq": {
"models": [
"llama-3.3-70b-versatile",
"llama-3.1-70b-versatile",
"llama-3.1-8b-instant",
"llama3-70b-8192",
"llama3-8b-8192",
"mixtral-8x7b-32768",
"gemma2-9b-it"
"meta-llama/llama-4-scout-17b-16e-instruct",
"openai/gpt-oss-120b",
"openai/gpt-oss-20b"
],
"recommended": "llama-3.3-70b-versatile",
"_note": "Not yet re-verified in 2026-04 refresh — kept from previous curation. Run the verifier with a Groq key to prune deprecated entries."
"_note": "Verified functionally 2026-07-14 with the Groq API (15 models discovered, 9 passed). Legacy llama-3.1-70b-versatile / llama3-70b-8192 / llama3-8b-8192 / mixtral-8x7b-32768 / gemma2-9b-it removed (retired upstream). llama-4-scout added (current-gen Llama 4, 0.47s). openai/gpt-oss-120b / gpt-oss-20b confirmed. Passing but excluded: allam-2-7b (Arabic-focused), qwen/qwen3-32b (Chinese-first, unreliable Spanish output), openai/gpt-oss-safeguard-20b (safety-classifier variant), groq/compound-mini (agentic system, wrong fit for notification translation)."
},
"gemini": {
"models": [
"gemini-flash-lite-latest",
"gemini-2.5-flash-lite",
"gemini-2.5-flash",
"gemini-3-flash-preview"
"gemini-3.1-flash-lite",
"gemini-3-flash-preview",
"gemini-3.5-flash"
],
"recommended": "gemini-2.5-flash-lite",
"_note": "flash-lite / flash pass the verifier consistently; pro variants reject thinkingBudget=0 and are overkill for notification translation anyway. 'latest' aliases (gemini-flash-latest, gemini-flash-lite-latest) are intentionally omitted because they resolved to different models across runs and produced timeouts in some regions.",
"_note": "Verified 2026-07-13. gemini-flash-lite-latest now passes consistently (1.6s) and is fastest, but gemini-2.5-flash-lite remains recommended because 'latest' aliases can drift over time. gemini-3.1-flash-lite is the stable successor to 3-flash-preview. Pro variants continue to reject thinkingBudget=0 and are overkill for notification translation.",
"_deprecated": ["gemini-2.0-flash", "gemini-2.0-flash-lite", "gemini-1.5-flash", "gemini-1.0-pro", "gemini-pro"]
},
@@ -36,21 +37,23 @@
"gpt-4.1",
"gpt-4o",
"gpt-5-chat-latest",
"gpt-5.4-nano",
"gpt-5.4-mini"
"gpt-5-nano"
],
"recommended": "gpt-4.1-nano",
"_note": "Reasoning models (o-series, gpt-5/5.1/5.2 non-chat variants) are supported by openai_provider.py via max_completion_tokens + reasoning_effort=minimal, but not listed here by default: their latency is higher than the chat models and they do not improve translation quality for notifications. Add specific reasoning IDs to this list only if a user explicitly wants them."
"_note": "Verified 2026-07-13. gpt-5.4-nano / gpt-5.4-mini removed (HTTP 400 — provider params rejected). gpt-5-nano added (2.0s, current-gen fast). Reasoning models (o-series, gpt-5/5.1/5.2 non-chat variants) are supported by openai_provider.py via max_completion_tokens + reasoning_effort=minimal, but not listed here: their latency is higher and they do not improve translation quality for notifications. Add specific reasoning IDs to this list only if a user explicitly wants them."
},
"anthropic": {
"models": [
"claude-3-5-haiku-latest",
"claude-3-5-sonnet-latest",
"claude-3-opus-latest"
"claude-haiku-4-5",
"claude-sonnet-5",
"claude-opus-4-8",
"claude-sonnet-4-6",
"claude-opus-4-6",
"claude-fable-5"
],
"recommended": "claude-3-5-haiku-latest",
"_note": "Not re-verified in 2026-04 refresh — kept from previous curation. Add claude-4.x / claude-4.5 / claude-4.6 / claude-4.7 variants after running the verifier with an Anthropic key."
"recommended": "claude-haiku-4-5",
"_note": "Verified 2026-07-13 with all 10 discovered models passing after aligning the verifier with anthropic_provider.py (temperature omitted — newest generations reject it with 'temperature is deprecated for this model'). Legacy claude-3-5-haiku-latest / claude-3-5-sonnet-latest / claude-3-opus-latest removed (deprecated upstream, not in the Models API). haiku-4-5 is the sweet spot for notification translation (3.6s, $1/$5 per MTok); sonnet-5 for slightly richer output (3.1s, $3/$15); opus-4-8 / fable-5 for demanding cases."
},
"openrouter": {
@@ -58,15 +61,16 @@
"meta-llama/llama-3.3-70b-instruct",
"meta-llama/llama-3.1-70b-instruct",
"meta-llama/llama-3.1-8b-instruct",
"anthropic/claude-3.5-haiku",
"anthropic/claude-3.5-sonnet",
"google/gemini-flash-1.5",
"meta-llama/llama-4-scout",
"anthropic/claude-haiku-4.5",
"anthropic/claude-sonnet-4.6",
"google/gemini-2.5-flash-lite",
"google/gemini-2.5-flash",
"openai/gpt-4o-mini",
"mistralai/mistral-7b-instruct",
"mistralai/mixtral-8x7b-instruct"
"mistralai/mistral-small-3.2-24b-instruct"
],
"recommended": "meta-llama/llama-3.3-70b-instruct",
"_note": "Not re-verified in 2026-04 refresh. google/gemini-flash-2.5-flash-lite was malformed in the previous entry and has been replaced with google/gemini-flash-1.5."
"_note": "Verified functionally 2026-07-14 with the OpenRouter API — all 10 curated candidates pass the Spanish-translation notification test. Fastest: llama-4-scout (0.51s), gemini-2.5-flash-lite (1.14s), gemini-2.5-flash (1.94s), llama-3.3-70b-instruct (2.29s), claude-haiku-4.5 (2.71s). Legacy anthropic/claude-3.5-* / google/gemini-flash-1.5 / mistralai/mistral-7b-instruct / mixtral-8x7b-instruct removed (GONE from catalog). Modern replacements added: llama-4-scout (Meta's current gen — dramatically fastest), claude-haiku-4.5 / claude-sonnet-4.6, gemini-2.5-flash / flash-lite, mistral-small-3.2-24b. recommended kept as llama-3.3-70b for capability/latency balance; llama-4-scout is a faster alternative worth considering as recommended after a broader release."
},
"ollama": {
+17 -3
View File
@@ -138,6 +138,12 @@ export async function fetchApi<T>(endpoint: string, options?: RequestInit): Prom
// return `{error: "..."}` on failure (e.g. /api/vms/<id>/control
// includes the pvesh stderr — telling the user "no space left on
// device" is infinitely more useful than the raw status text).
//
// We also attach the FULL parsed JSON body to the thrown Error
// as `.body` so callers that want the optional `details` /
// `suggestion` fields (e.g. /api/node/metrics) can render them
// without re-fetching. Callers that just read `err.message`
// keep working exactly as before.
try {
const ct = response.headers.get("content-type") || ""
if (ct.includes("application/json")) {
@@ -145,14 +151,22 @@ export async function fetchApi<T>(endpoint: string, options?: RequestInit): Prom
const detail =
(body && (body.error || body.message)) || ""
if (detail) {
throw new Error(detail)
const e: Error & { body?: unknown; status?: number } = new Error(detail)
e.body = body
e.status = response.status
throw e
}
}
} catch (parseErr) {
if (parseErr instanceof Error && parseErr.message.includes("API request failed")) {
// Backend-supplied detail (the explicit `throw new Error(detail)`
// above) MUST propagate so the UI shows "path does not exist…"
// instead of the generic "API request failed: 400 BAD REQUEST".
// Only swallow when the JSON itself failed to parse — that's a
// real SyntaxError and falling through to the generic message
// is the right behaviour there.
if (!(parseErr instanceof SyntaxError)) {
throw parseErr
}
// JSON parse failed — fall through to the generic message.
}
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
}
+30
View File
@@ -0,0 +1,30 @@
// Shared classifier for physical-disk type. Lives here because the
// Storage page and the Hardware page used to ship their own copies
// and silently drifted — old SSDs (e.g. OCZ-SOLID2) that don't expose
// a SMART rotation rate fell through Hardware's HDD-as-default branch
// and got mislabelled, while the Storage page got it right.
//
// Backend convention for `rotation_rate`:
// undefined / null / 0 → SSD (no platters reported)
// -1 → HDD detected via /sys rotational flag,
// but the drive doesn't expose RPM
// > 0 → HDD with known RPM
// string "Solid State" → SSD (smartctl wording on a few vendors)
export type DiskType = "NVMe" | "SSD" | "HDD"
export function getDiskType(
diskName: string,
rotationRate: number | string | null | undefined,
): DiskType {
if (diskName.startsWith("nvme")) return "NVMe"
if (rotationRate === -1) return "HDD"
if (typeof rotationRate === "string") {
if (rotationRate.includes("Solid State")) return "SSD"
const parsed = Number.parseInt(rotationRate, 10)
if (Number.isNaN(parsed) || parsed === 0) return "SSD"
return "HDD"
}
if (rotationRate == null || rotationRate === 0) return "SSD"
return "HDD"
}
+48
View File
@@ -0,0 +1,48 @@
// Shared usage-bar palette for storage capacity widgets. Extracted
// so the Storage page (overview cards + per-storage rows) and the
// Backups page (Available Archives → per-archive capacity bar) flag
// a full datastore with the same colour. Previously the Backups bar
// was hard-coded to blue, so a 100%-full PBS-Cloud appeared in red
// on Storage and in blue on Backups — same datastore, two different
// signals.
//
// Thresholds: < 75 % blue (normal — no alert), 7589 % amber,
// ≥ 90 % red. Matches the Storage page palette: the normal state
// stays on the project's brand blue and only switches to amber/red
// when the operator should look at it. Green is reserved for OK
// signals where green has meaning (SMART status, wear level), not
// for ambient bars.
export type UsageBarColor = {
/** Inline `background` value for SVG / style={} consumers. */
hex: string
/** Tailwind `bg-*` class for div consumers. */
bgClass: string
/** Tailwind `text-*` class for "Free" / counters that should
* share the urgency signal. */
textClass: string
}
const BLUE: UsageBarColor = {
hex: "#3b82f6",
bgClass: "bg-blue-500",
// No text-blue override for the "Free" counter — at normal usage
// the foreground colour reads better than tinted text.
textClass: "",
}
const AMBER: UsageBarColor = {
hex: "#f59e0b",
bgClass: "bg-amber-500",
textClass: "text-amber-400",
}
const RED: UsageBarColor = {
hex: "#ef4444",
bgClass: "bg-red-500",
textClass: "text-red-400",
}
export function getStorageUsageColor(percent: number): UsageBarColor {
if (percent >= 90) return RED
if (percent >= 75) return AMBER
return BLUE
}
+17
View File
@@ -19,3 +19,20 @@ export function formatStorage(sizeInGB: number): string {
return `${tb % 1 === 0 ? tb.toFixed(0) : tb.toFixed(1)} TB`
}
}
// Byte-aware formatter. Scales B → KB → MB → GB → TB. Use when the
// raw value comes in bytes (log file sizes from os.path.getsize(),
// PBS / Borg datastore capacity reported by the backend in bytes).
// The Backups page used to ship its own copy that capped at GB, so a
// 7 TB datastore showed up as "7311.55 GB" — extracted here so it
// can't drift again.
export function formatBytes(n: number): string {
if (!Number.isFinite(n) || n < 0) return "—"
if (n < 1024) return `${n} B`
if (n < 1024 * 1024) return `${(n / 1024).toFixed(1)} KB`
if (n < 1024 * 1024 * 1024) return `${(n / (1024 * 1024)).toFixed(1)} MB`
if (n < 1024 * 1024 * 1024 * 1024) {
return `${(n / (1024 * 1024 * 1024)).toFixed(2)} GB`
}
return `${(n / (1024 * 1024 * 1024 * 1024)).toFixed(2)} TB`
}
+2 -2
View File
@@ -1,12 +1,12 @@
{
"name": "ProxMenux-Monitor",
"version": "1.2.2",
"version": "1.2.2.2-beta",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "ProxMenux-Monitor",
"version": "1.2.2",
"version": "1.2.2.2-beta",
"dependencies": {
"@hookform/resolvers": "^3.10.0",
"@radix-ui/react-accordion": "1.2.2",
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "ProxMenux-Monitor",
"version": "1.2.2.1-beta",
"version": "1.2.4",
"description": "Proxmox System Monitoring Dashboard",
"private": true,
"scripts": {
Binary file not shown.

After

Width:  |  Height:  |  Size: 9.3 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 16 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 25 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 14 KiB

+18 -3
View File
@@ -3,14 +3,29 @@
"short_name": "ProxMenux",
"description": "Proxmox System Dashboard and Monitor",
"start_url": "/",
"scope": "/",
"display": "standalone",
"orientation": "any",
"background_color": "#2b2f36",
"theme_color": "#2b2f36",
"icons": [
{
"src": "/images/proxmenux-logo.png",
"sizes": "256x256",
"type": "image/png"
"src": "/icons/icon-192.png",
"sizes": "192x192",
"type": "image/png",
"purpose": "any"
},
{
"src": "/icons/icon-512.png",
"sizes": "512x512",
"type": "image/png",
"purpose": "any"
},
{
"src": "/icons/icon-maskable-512.png",
"sizes": "512x512",
"type": "image/png",
"purpose": "maskable"
}
]
}
+40
View File
@@ -0,0 +1,40 @@
// ==========================================================
// ProxMenux Monitor — Service Worker
// ==========================================================
// Minimal SW whose only job is to make Chrome (Android) treat
// the Monitor as an installable PWA. The Monitor lives on the
// operator's LAN, hits a self-signed HTTPS endpoint and has
// ZERO offline value (every page calls /api/* over the wire),
// so we do NOT cache pages or API responses — caching them
// would only cause stale-data bugs after an AppImage update.
//
// The install / activate handlers clean up old SW caches from
// previous Monitor versions so a beta-to-stable upgrade never
// strands the browser on a cached old shell.
// ==========================================================
const SW_VERSION = 'proxmenux-monitor-v1';
self.addEventListener('install', (event) => {
// Take over as soon as installed; no skipWaiting handshake.
self.skipWaiting();
});
self.addEventListener('activate', (event) => {
event.waitUntil((async () => {
// Wipe any cache name that isn't ours — survives renames /
// version bumps without piling up stale entries.
const names = await caches.keys();
await Promise.all(
names.filter((n) => n !== SW_VERSION).map((n) => caches.delete(n))
);
await self.clients.claim();
})());
});
// Network-only fetch. The SW exists so Chrome marks the site as
// installable; we deliberately do not serve cached responses.
self.addEventListener('fetch', (event) => {
// Let the browser handle it normally — no respondWith → no cache.
return;
});
@@ -15,12 +15,21 @@ class AnthropicProvider(AIProvider):
API_URL = "https://api.anthropic.com/v1/messages"
API_VERSION = "2023-06-01"
# Known stable model aliases (Anthropic doesn't have a public models list API)
# These use "-latest" which auto-updates to the newest version
# Anthropic model aliases that resolve to pinned snapshots (per docs
# at platform.claude.com/docs/en/docs/about-claude/models). Kept as a
# runtime fallback for the `/api/notifications/provider-models`
# intersection with `verified_ai_models.json`; the JSON is the
# authoritative UI-facing list. Refreshed 2026-07-13 to drop the
# deprecated claude-3-5-* aliases (retired upstream, no longer in the
# Models API response) and add the current-generation IDs. Ordered
# cheapest → most capable so the first entry is a safe default.
KNOWN_MODELS = [
"claude-3-5-haiku-latest",
"claude-3-5-sonnet-latest",
"claude-3-opus-latest",
"claude-haiku-4-5",
"claude-sonnet-5",
"claude-sonnet-4-6",
"claude-opus-4-8",
"claude-opus-4-6",
"claude-fable-5",
]
def list_models(self) -> List[str]:
+55 -132
View File
@@ -166,129 +166,54 @@ else
echo "⚠️ config directory not found"
fi
echo "📋 Adding translation support..."
cat > "$APP_DIR/usr/bin/translate_cli.py" << 'PYEOF'
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
ProxMenux translate CLI
stdin JSON -> {"text":"...", "dest_lang":"es", "context":"...", "cache_file":"/usr/local/share/proxmenux/cache.json"}
stdout JSON -> {"success":true,"text":"..."} or {"success":false,"error":"..."}
"""
import sys, json, re
from pathlib import Path
# Translation handling lives in scripts/utils.sh now. It reads
# /usr/local/share/proxmenux/lang/<lang>.json (pre-built by the
# build_translation_cache.py CI job) and falls back to the English
# source string on miss. The Monitor AppImage no longer ships the
# runtime translate_cli.py — the JSON files belong to the host install,
# not to the Flask dashboard.
# Ensure embedded site-packages are discoverable
HERE = Path(__file__).resolve().parents[2] # .../AppDir
DIST = HERE / "usr" / "lib" / "python3" / "dist-packages"
SITE = HERE / "usr" / "lib" / "python3" / "site-packages"
for p in (str(DIST), str(SITE)):
if p not in sys.path:
sys.path.insert(0, p)
# Python 3.13 compat: inline 'cgi' shim
try:
import cgi
except Exception:
import types, html
def _parse_header(value: str):
value = str(value or "")
parts = [p.strip() for p in value.split(";")]
if not parts:
return "", {}
key = parts[0].lower()
params = {}
for item in parts[1:]:
if not item:
continue
if "=" in item:
k, v = item.split("=", 1)
k = k.strip().lower()
v = v.strip().strip('"').strip("'")
params[k] = v
else:
params[item.strip().lower()] = ""
return key, params
cgi = types.SimpleNamespace(parse_header=_parse_header, escape=html.escape)
try:
from googletrans import Translator
except Exception as e:
print(json.dumps({"success": False, "error": f"ImportError: {e}"}))
sys.exit(0)
def load_json_stdin():
try:
return json.load(sys.stdin)
except Exception as e:
print(json.dumps({"success": False, "error": f"Invalid JSON input: {e}"}))
sys.exit(0)
def ensure_cache(path: Path):
try:
path.parent.mkdir(parents=True, exist_ok=True)
if not path.exists():
path.write_text("{}", encoding="utf-8")
json.loads(path.read_text(encoding="utf-8") or "{}")
except Exception:
path.write_text("{}", encoding="utf-8")
def read_cache(path: Path):
try:
return json.loads(path.read_text(encoding="utf-8") or "{}")
except Exception:
return {}
def write_cache(path: Path, cache: dict):
tmp = path.with_suffix(".tmp")
tmp.write_text(json.dumps(cache, ensure_ascii=False), encoding="utf-8")
tmp.replace(path)
def clean_translated(s: str) -> str:
s = re.sub(r'^.*?(Translate:|Traducir:|Traduire:|Übersetzen:|Tradurre:|Traduzir:|翻译:|翻訳:)', '', s, flags=re.IGNORECASE | re.DOTALL).strip()
s = re.sub(r'^.*?(Context:|Contexto:|Contexte:|Kontext:|Contesto:|上下文:|コンテキスト:).*?:', '', s, flags=re.IGNORECASE | re.DOTALL).strip()
return s.strip()
def main():
req = load_json_stdin()
text = req.get("text", "")
dest = req.get("dest_lang", "en") or "en"
context = req.get("context", "")
cache_file = Path(req.get("cache_file", "")) if req.get("cache_file") else None
if dest == "en":
print(json.dumps({"success": True, "text": text}))
return
cache = {}
if cache_file:
ensure_cache(cache_file)
cache = read_cache(cache_file)
if text in cache and (dest in cache[text] or "notranslate" in cache[text]):
found = cache[text].get(dest) or cache[text].get("notranslate")
print(json.dumps({"success": True, "text": found}))
return
try:
full = (context + " " + text).strip() if context else text
tr = Translator()
result = tr.translate(full, dest=dest).text
result = clean_translated(result)
if cache_file:
cache.setdefault(text, {})
cache[text][dest] = result
write_cache(cache_file, cache)
print(json.dumps({"success": True, "text": result}))
except Exception as e:
print(json.dumps({"success": False, "error": str(e)}))
if __name__ == "__main__":
main()
PYEOF
chmod +x "$APP_DIR/usr/bin/translate_cli.py"
# ── Borg standalone binary ─────────────────────────────────────────
# Ship the official borg standalone binary inside the AppImage so the
# host-backup / restore workflows can run without an internet round-trip
# at install time. Pinned to the same version that proxmenux's
# hb_ensure_borg used to download on demand — kept in lockstep so both
# code paths see the same version semantics. SHA256 is the upstream
# release checksum; bump both together.
BORG_VERSION="1.2.8"
BORG_URL="https://github.com/borgbackup/borg/releases/download/${BORG_VERSION}/borg-linux64"
BORG_SHA256="cfa50fb704a93d3a4fa258120966345fddb394f960dca7c47fcb774d0172f40b"
echo "📦 Downloading borg ${BORG_VERSION} into AppImage..."
BORG_TARGET="$APP_DIR/usr/bin/borg"
# GitHub releases serve borg-linux64 via a 302 redirect to a signed
# release-assets.githubusercontent.com URL. wget -qO silently dropped
# the redirect once during the 2026-06-15 build, killing the AppImage
# pipeline. curl -L --retry 3 is more robust and falls back to wget
# only when curl is missing.
if command -v curl >/dev/null 2>&1; then
DOWNLOAD_OK=0
if curl -sSL --retry 3 --retry-delay 2 --max-time 120 -o "$BORG_TARGET" "$BORG_URL"; then
DOWNLOAD_OK=1
fi
else
DOWNLOAD_OK=0
if wget -qO "$BORG_TARGET" "$BORG_URL"; then
DOWNLOAD_OK=1
fi
fi
if [ "$DOWNLOAD_OK" = "1" ]; then
if echo "${BORG_SHA256} ${BORG_TARGET}" | sha256sum -c - >/dev/null 2>&1; then
chmod +x "$BORG_TARGET"
echo "✅ borg ${BORG_VERSION} bundled (sha256 verified)"
else
echo "❌ borg sha256 verification failed — removing"
rm -f "$BORG_TARGET"
exit 1
fi
else
echo "❌ borg download failed from $BORG_URL"
exit 1
fi
# Copy Next.js build
echo "📋 Copying web dashboard..."
@@ -332,7 +257,7 @@ cat > "$APP_DIR/proxmenux-monitor.desktop" << EOF
[Desktop Entry]
Type=Application
Name=ProxMenux Monitor
Comment=Proxmox System Monitoring Dashboard with Translation Support
Comment=Proxmox System Monitoring Dashboard
Exec=AppRun
Icon=proxmenux-monitor
Categories=System;Monitor;
@@ -361,14 +286,12 @@ if [ -f "$APP_DIR/proxmenux-monitor.png" ]; then
fi
echo "📦 Installing Python dependencies..."
# Phase 1: Install googletrans with its old dependencies
pip3 install --target "$APP_DIR/usr/lib/python3/dist-packages" \
googletrans==4.0.0-rc1 \
httpx==0.13.3 \
httpcore==0.9.1 \
h11==0.9.0 || true
# Phase 2: Install modern Flask/WebSocket dependencies (will upgrade h11 and related packages)
# Flask/WebSocket dependencies for the Monitor dashboard. The previous
# Phase-1 (googletrans==4.0.0-rc1 + httpx 0.13.3 + httpcore 0.9.1 +
# h11 0.9.0) is gone — translation is now a static-lookup feature on
# the host, so the AppImage no longer needs any runtime translator.
# Removing those pins also unblocks the h11>=0.14.0 family without the
# conflict workaround we used to ship.
# Note: cryptography removed due to Python version compatibility issues (PyO3 modules)
pip3 install --target "$APP_DIR/usr/lib/python3/dist-packages" --upgrade --no-deps \
flask \
@@ -380,7 +303,7 @@ pip3 install --target "$APP_DIR/usr/lib/python3/dist-packages" --upgrade --no-de
segno \
beautifulsoup4
# Phase 3: Install WebSocket with newer h11
# WebSocket with modern h11 (no need for the legacy pin anymore)
pip3 install --target "$APP_DIR/usr/lib/python3/dist-packages" --upgrade \
h11>=0.14.0 \
wsproto>=1.2.0 \
+95 -11
View File
@@ -237,22 +237,64 @@ def _list_target_disks() -> list[str]:
return fresh
def _is_disk_usb(disk_name: str) -> bool:
"""True if the disk sits behind a USB bus, checked via the resolved
sysfs device path. USB-NVMe bridges (ASMedia, JMicron, Realtek) and
plain USB-HDDs both report `/sys/block/<disk>/removable = 0`, so the
older removable-flag heuristic missed them and the temperature
poller never tried the snt* driver variants that are the only way
to reach the NVMe controller behind those bridges."""
try:
base = disk_name[5:] if disk_name.startswith('/dev/') else disk_name
real = os.path.realpath(f'/sys/block/{base}')
return any(seg.startswith('usb') and (len(seg) == 3 or seg[3:].isdigit())
for seg in real.split('/'))
except Exception:
return False
def _smartctl_cmd_for(disk_name: str, probe: str) -> list[str]:
"""Build the smartctl invocation for a given probe key."""
cmd = ["smartctl", "-A", "-j"]
"""Build the smartctl invocation for a given probe key.
`-n standby` makes smartctl exit immediately with code 2 (no disk
I/O) when the drive is already in standby. Without it, this
once-a-minute poller was spinning HDDs back up on every cycle,
breaking NAS / SnapRAID setups that rely on hdparm-driven spin-down
(issue #232).
"""
cmd = ["smartctl", "-n", "standby", "-A", "-j"]
if probe != "auto":
cmd.extend(["-d", probe])
cmd.append(f"/dev/{disk_name}")
return cmd
# Sentinel returned by `_try_probe` when the drive is in standby. Distinct
# from `None` (read failure / no temperature attribute), so the caller
# can keep the last known reading instead of marking the disk as failing.
_STANDBY = "standby"
def _try_probe(disk_name: str, probe: str) -> Optional[float]:
"""Run a single smartctl invocation and parse the temperature."""
"""Run a single smartctl invocation and parse the temperature.
Returns:
* a float current temperature in °C.
* the string ``_STANDBY`` drive is in standby, NOT read.
* ``None`` read failed for any other reason.
"""
try:
proc = subprocess.run(
_smartctl_cmd_for(disk_name, probe),
capture_output=True, text=True, timeout=_SMARTCTL_TIMEOUT,
)
# `-n standby` makes smartctl exit with code 2 when the drive is
# parked. We must not treat that as a read failure (would trigger
# the backoff and stop polling that drive forever) — surface it
# as the dedicated _STANDBY sentinel so the caller skips the
# update cleanly.
if proc.returncode == 2:
return _STANDBY # type: ignore[return-value]
# smartctl returns non-zero on warnings (bit 0x40 etc.) even when
# JSON is fully populated. Don't gate on returncode — parse the
# body regardless.
@@ -264,8 +306,24 @@ def _try_probe(disk_name: str, probe: str) -> Optional[float]:
return None
# Disks that returned "standby" on their last poll. Used by the
# /api/storage/disks endpoint to render a Standby badge so the operator
# understands why the temperature graph for that drive is frozen — the
# disk really is parked, not the monitor that's broken.
_standby_state: dict[str, float] = {} # disk_name -> last-seen timestamp
_STANDBY_TTL = 600 # treat as stale after 10 min of no observation
def is_disk_in_standby(disk_name: str) -> bool:
"""True if our last smartctl poll for this disk hit a standby spindle.
Falls back to False when the cached observation is older than the
TTL the drive may have woken up between polls."""
ts = _standby_state.get(disk_name)
return ts is not None and (time.time() - ts) < _STANDBY_TTL
def _read_temperature(disk_name: str) -> Optional[float]:
"""Pull the current temperature from ``smartctl -A -j``.
"""Pull the current temperature from ``smartctl -n standby -A -j``.
Caching strategy:
* If we've previously found a working probe for this disk we go
@@ -275,6 +333,9 @@ def _read_temperature(disk_name: str) -> Optional[float]:
and update the cache with whatever does work.
* Disks that never report a temperature get rate-limited via the
backoff table so we don't smartctl them every minute forever.
* Disks in standby return ``None`` but DON'T count toward the
failure backoff they're not broken, they're just parked.
The standby state is recorded so the UI can show a badge.
"""
now = time.time()
@@ -285,10 +346,23 @@ def _read_temperature(disk_name: str) -> Optional[float]:
if retry_at > now:
return None
def _handle(result):
"""Clear failure state + record standby observation. Returns the
numeric temperature if the result is one (else None / standby)."""
if result == _STANDBY:
_standby_state[disk_name] = time.time()
return _STANDBY
if isinstance(result, (int, float)) and result > 0:
_standby_state.pop(disk_name, None)
return result
return None
# Fast path: cached probe.
if cached_probe is not None:
temp = _try_probe(disk_name, cached_probe)
if temp is not None and temp > 0:
temp = _handle(_try_probe(disk_name, cached_probe))
if temp == _STANDBY:
return None # parked — skip update, don't penalise
if temp is not None:
with _cache_lock:
_disk_fail_counts.pop(disk_name, None)
_disk_fail_backoff.pop(disk_name, None)
@@ -296,19 +370,29 @@ def _read_temperature(disk_name: str) -> Optional[float]:
# Cached probe stopped working — fall through and re-detect.
# Slow path: try every probe and remember the first one that works.
for probe in ("auto", "nvme", "ata", "sat"):
# For USB-attached disks we prepend the three snt* driver variants —
# USB-NVMe bridges (ASMedia / JMicron / Realtek) don't answer the
# plain probes with real SMART; only snt* passes through to the NVMe
# controller so temperature actually comes back. Non-USB disks skip
# them, so this adds zero overhead on internal drives.
probes: tuple[str, ...] = ("auto", "nvme", "ata", "sat")
if _is_disk_usb(disk_name):
probes = ("sntasmedia", "sntjmicron", "sntrealtek") + probes
for probe in probes:
if probe == cached_probe:
continue # already tried above
temp = _try_probe(disk_name, probe)
if temp is not None and temp > 0:
temp = _handle(_try_probe(disk_name, probe))
if temp == _STANDBY:
return None
if temp is not None:
with _cache_lock:
_disk_probe_cache[disk_name] = probe
_disk_fail_counts.pop(disk_name, None)
_disk_fail_backoff.pop(disk_name, None)
return temp
# All probes failed. Bump the failure counter and trip the backoff
# if we've crossed the threshold.
# All probes failed (none returned a temperature OR standby). Bump
# the failure counter and trip the backoff if threshold crossed.
with _cache_lock:
n = _disk_fail_counts.get(disk_name, 0) + 1
_disk_fail_counts[disk_name] = n
+19 -1
View File
@@ -101,7 +101,14 @@ def acknowledge_error():
'security': 'security_check',
'temperature': 'cpu_check',
'network': 'network_check',
# Both 'disks' (kept for compat) and 'storage' land on the
# same cache — the two categories share `storage_check` in
# health_monitor. Without the 'storage' entry, dismissing
# a storage_unavailable / mount_stale / lxc_mount_low
# error persisted but never invalidated the cache, so the
# error stayed visible in the next fetch.
'disks': 'storage_check',
'storage': 'storage_check',
'vms': 'vms_check',
}
cache_key = cache_key_map.get(category)
@@ -223,14 +230,25 @@ def get_full_health():
Get complete health data in a single request: detailed status + active errors + dismissed.
Uses background-cached results if fresh (< 6 min) for instant response,
otherwise runs a fresh check.
?refresh=1 busts the background + per-check caches for updates/services/
security before returning, so an event that just changed underlying state
(Update Now finished, dismiss action) sees the new value immediately
instead of waiting for the next polling tick.
"""
import time as _time
try:
if request.args.get('refresh') == '1':
for ck in ('updates_check', 'pve_services', 'security_check',
'_bg_detailed', '_bg_overall', 'overall_health'):
health_monitor.last_check_times.pop(ck, None)
health_monitor.cached_results.pop(ck, None)
# Try to use the background-cached detailed result for instant response
bg_key = '_bg_detailed'
bg_last = health_monitor.last_check_times.get(bg_key, 0)
bg_age = _time.time() - bg_last
if bg_age < 360 and bg_key in health_monitor.cached_results:
# Use cached result (at most ~5 min old)
details = health_monitor.cached_results[bg_key]
+284 -12
View File
@@ -214,6 +214,53 @@ def _is_loopback_addr(value: str) -> bool:
return value == 'localhost'
def _is_own_host_ip(value: str) -> bool:
"""Return True when ``value`` is loopback OR an IP bound to any local iface.
``_pve_webhook_url()`` may register a URL that resolves to the host's
LAN/VPN IP when SSL is on and a hostname cert is loaded (issue #239).
In that case PVE POSTs to e.g. ``https://<fqdn>:8008`` and on Linux
the connection is routed to the local interface holding that IP; the
Flask socket sees the peer as the interface IP, NOT ``127.0.0.1``. The
request is still coming from THIS host, so the loopback trust path
should extend to any of this host's own interface IPs (Tailscale/Zerotier
CGNAT, WireGuard, LAN, IPv6 GUA). Without this, PVE hits the layer 3
``X-ProxMenux-Timestamp`` check a header PVE cannot inject dynamically
and every notification target test returns ``401 missing_timestamp``.
IMPORTANT: Flask bound to ``*:8008`` (dual-stack) reports IPv4 peers
in v4-mapped IPv6 form (``::ffff:192.168.0.55``). psutil reports the
same interface as plain IPv4 (``192.168.0.55``). Without unmapping,
literal comparison fails and the fix effectively does nothing in
production reproduced end-to-end on 192.168.0.55: passing the raw
literal returned True, but ``::ffff:192.168.0.55`` returned False,
which is what Flask actually hands us at runtime.
"""
if _is_loopback_addr(value):
return True
try:
import ipaddress
import socket
import psutil
addr = ipaddress.ip_address(value)
mapped = getattr(addr, 'ipv4_mapped', None)
if mapped is not None:
addr = mapped
client = addr.compressed
for _iface, addrs in psutil.net_if_addrs().items():
for a in addrs:
if a.family in (socket.AF_INET, socket.AF_INET6):
ip_str = a.address.split('%')[0] # strip IPv6 zone id
try:
if ipaddress.ip_address(ip_str).compressed == client:
return True
except ValueError:
continue
except Exception:
pass
return False
def _validate_event_type(value: str) -> bool:
return isinstance(value, str) and bool(_EVENT_TYPE_RE.match(value))
@@ -266,6 +313,74 @@ def save_notification_settings():
return jsonify({'error': f'Internal error ({type(e).__name__})'}), 500
@notification_bp.route('/api/notifications/reveal-secret', methods=['POST'])
@require_auth
def reveal_notification_secret():
"""Return one sensitive config value in cleartext.
Backs the "eye" toggle in the Settings UI. The settings GET masks
every entry in SENSITIVE_KEYS with `'************'` so the secret
never leaves the server just because someone loaded the page; this
endpoint lets an authenticated operator explicitly request the
real value for a single key when they need to inspect it.
Body schema (one of):
{"ai_provider": "groq" | "anthropic" | }
{"channel": "telegram", "field": "bot_token"}
{"key": "webhook_secret"}
Returns ``{"value": "<cleartext>"}`` or 404 when nothing is stored.
400 on a malformed request, 403 if the requested key isn't on the
whitelist (i.e. the UI is asking for something we never agreed to
expose defence against a future bug binding the eye to a
non-secret field).
"""
try:
from notification_manager import SENSITIVE_KEYS
nm = notification_manager # singleton already imported at module top
payload = request.get_json(silent=True) or {}
# Resolve the requested config key from the three accepted
# body shapes. Whitelist-checked below before we touch the DB.
cfg_key = None
provider = (payload.get('ai_provider') or '').strip().lower()
channel = (payload.get('channel') or '').strip().lower()
field = (payload.get('field') or '').strip().lower()
raw_key = (payload.get('key') or '').strip()
if provider:
cfg_key = f'ai_api_key_{provider}'
elif channel and field:
cfg_key = f'{channel}.{field}'
elif raw_key:
cfg_key = raw_key
if not cfg_key:
return jsonify({'error': 'missing_target'}), 400
# Whitelist enforcement — never reveal a value the rest of the
# system doesn't recognise as a secret. Stops a hypothetical
# future bug where the UI passes a non-secret config key and
# we hand back something we shouldn't.
if cfg_key not in SENSITIVE_KEYS:
return jsonify({'error': 'forbidden_key'}), 403
# Read the raw value through notification_manager (loads + caches
# the config). Mask placeholder is the same string the GET would
# return — treat that as "not stored" so we don't echo it back.
if not nm._config:
nm._load_config()
value = nm._config.get(cfg_key, '') or ''
if not value or value == '************':
return jsonify({'value': ''}), 200
return jsonify({'value': value}), 200
except Exception as e:
print(f"[notification_routes] {request.path} failed: {type(e).__name__}: {e}")
return jsonify({'error': f'Internal error ({type(e).__name__})'}), 500
@notification_bp.route('/api/notifications/test', methods=['POST'])
@require_auth
def test_notification():
@@ -721,23 +836,108 @@ _PVE_OUR_HEADERS = {
}
def _pve_webhook_url() -> str:
"""Return http:// or https:// based on the current SSL config.
def _ssl_cert_hostname(cert_path: str) -> str:
"""Pull the most useful hostname out of an x509 cert.
Hardcoded `http://...` previously broke webhook delivery whenever the
user enabled SSL Flask only listened on HTTPS, so PVE got connection
refused and notifications stopped. Issue #194. PVE may still need
`update-ca-certificates` if the cert is self-signed; that's a doc
step on the user side.
Preference order: first DNS SAN CN. Returns '' on any failure.
Used to build a webhook URL that won't fail PVE's TLS verification
(issue #239 — PVE has no `--insecure` flag and the user's ACME cert
is bound to a hostname, not to `127.0.0.1`).
"""
try:
import subprocess
out = subprocess.run(
['openssl', 'x509', '-in', cert_path, '-noout',
'-ext', 'subjectAltName', '-subject'],
capture_output=True, text=True, timeout=5,
)
if out.returncode != 0:
return ''
text = out.stdout or ''
# SAN line example: " DNS:pve.example.com, DNS:pve, IP Address:..."
import re
for line in text.splitlines():
m = re.search(r'DNS:([A-Za-z0-9.\-]+)', line)
if m:
return m.group(1)
# CN fallback. "subject= CN = pve.example.com" or "...CN=pve.example.com"
m = re.search(r'CN\s*=\s*([A-Za-z0-9.\-]+)', text)
if m:
return m.group(1)
except Exception:
pass
return ''
def _hostname_resolves_locally(hostname: str) -> bool:
"""True when `hostname` resolves to one of this host's own IPs.
Anti-misconfig: refuse to build a webhook URL that points
elsewhere (a stale DNS entry pointing at the previous host, a CN
that names a different node in the cluster, etc.). PVE delivers
webhooks from the same node, so the URL has to round-trip to
ourselves.
"""
try:
import socket
import ipaddress
target_ips = set()
for info in socket.getaddrinfo(hostname, None):
ip = info[4][0]
target_ips.add(ipaddress.ip_address(ip).compressed)
# Collect our own IPs from /proc/net/fib_trie isn't portable; use
# psutil if available, otherwise fall back to socket on the
# hostname itself.
local_ips = {'127.0.0.1', '::1'}
try:
import psutil
for _iface, addrs in psutil.net_if_addrs().items():
for a in addrs:
if a.family in (socket.AF_INET, socket.AF_INET6):
local_ips.add(ipaddress.ip_address(a.address.split('%')[0]).compressed)
except Exception:
pass
return bool(target_ips & local_ips)
except Exception:
return False
def _pve_webhook_url() -> str:
"""Return the URL we register with PVE as our webhook target.
Three branches:
1. SSL off http://127.0.0.1:8008 (always works, no cert).
2. SSL on + cert hostname extractable and resolves locally
https://<cert-hostname>:8008. This is what PVE's TLS layer
actually validates against. Without this, PVE rejects the
self/ACME cert with "IP address mismatch" (issue #239).
3. SSL on but hostname extraction/check failed fall back to
https://127.0.0.1:8008 and accept that the user may still
hit the cert-mismatch error. We log so the operator can
diagnose. Better than silently emitting a wrong URL.
"""
try:
from auth_manager import load_ssl_config
cfg = load_ssl_config() or {}
if cfg.get('enabled'):
return 'https://127.0.0.1:8008/api/notifications/webhook'
if not cfg.get('enabled'):
return 'http://127.0.0.1:8008/api/notifications/webhook'
cert_path = cfg.get('cert_path') or ''
if cert_path:
host = _ssl_cert_hostname(cert_path)
if host and _hostname_resolves_locally(host):
return f'https://{host}:8008/api/notifications/webhook'
if host:
print(
f"[ProxMenux] webhook URL fallback to 127.0.0.1: "
f"cert hostname '{host}' does not resolve to a local "
f"IP — PVE will likely report a TLS verification "
f"error. Fix by ensuring the FQDN resolves on this "
f"host (e.g. /etc/hosts entry)."
)
return 'https://127.0.0.1:8008/api/notifications/webhook'
except Exception:
pass
return 'http://127.0.0.1:8008/api/notifications/webhook'
return 'http://127.0.0.1:8008/api/notifications/webhook'
# Backward-compat alias for callers that read this at import time. Most
@@ -1254,7 +1454,10 @@ def proxmox_webhook():
_reject = lambda code, error, status: (jsonify({'accepted': False, 'error': error}), status)
client_ip = request.remote_addr or ''
is_localhost = _is_loopback_addr(client_ip)
# Trust loopback AND any IP bound to a local interface — see
# `_is_own_host_ip` for the FQDN/CGNAT rationale. Layer 1 rate
# limiting still applies to every request.
is_localhost = _is_own_host_ip(client_ip)
# CSRF defence-in-depth: reject `application/x-www-form-urlencoded`
# bodies. PVE always sends `application/json`; form-encoded bodies
@@ -1407,3 +1610,72 @@ def internal_shutdown_event():
return jsonify({'success': True, 'event_type': event_type}), 200
except Exception as e:
return jsonify({'error': 'internal_error', 'detail': str(e)}), 500
# ─── Internal Restore Event Endpoint ─────────────────────────────
@notification_bp.route('/api/internal/restore-event', methods=['POST'])
def internal_restore_event():
"""
Internal endpoint called by apply_cluster_postboot.sh when the post-boot
dispatcher finishes. Tells the user the backgrounded restore tasks
(DKMS compile, apt installs, cluster apply, ...) are done so commands
like nvidia-smi now work.
Only accepts requests from localhost (127.0.0.1) for security.
"""
# Flask bound to 0.0.0.0 receives loopback connections as the
# IPv4-mapped IPv6 form (`::ffff:127.0.0.1`) on dual-stack hosts,
# not as bare `127.0.0.1`. Unwrap the mapped IPv4 before asking the
# stdlib classifier so every loopback representation (v4, v6,
# IPv4-mapped) is accepted.
remote_addr = request.remote_addr or ''
try:
import ipaddress
addr = ipaddress.ip_address(remote_addr.split('%')[0])
if isinstance(addr, ipaddress.IPv6Address) and addr.ipv4_mapped is not None:
addr = addr.ipv4_mapped
is_loopback = addr.is_loopback
except (ValueError, TypeError):
is_loopback = remote_addr in ('127.0.0.1', '::1', 'localhost')
if not is_loopback:
return jsonify({'error': 'forbidden', 'detail': 'localhost only'}), 403
try:
data = request.get_json(silent=True) or {}
hostname = data.get('hostname', 'unknown')
guests = data.get('guests', '0')
stubs = data.get('stubs', '0')
stale_nodes = data.get('stale_nodes', '0')
components = data.get('components', 'none')
duration = data.get('duration', 'unknown')
# Boot sanity-check warnings surfaced by apply_cluster_postboot.sh.
# Empty on a clean restore; populated on the cross-version path
# when the sanity check found something the operator should know
# about (missing /lib/modules for the default kernel, no ESP
# configured, dangling /vmlinuz, ...).
warnings = data.get('warnings', '').strip()
severity = 'WARNING' if warnings else 'INFO'
warnings_block = f'\n⚠️ Boot sanity: {warnings}\n' if warnings else ''
notification_manager.emit_event(
event_type='system_restore_completed',
severity=severity,
data={
'hostname': hostname,
'guests': guests,
'stubs': stubs,
'stale_nodes': stale_nodes,
'components': components,
'duration': duration,
'warnings': warnings,
'warnings_block': warnings_block,
},
source='proxmenux',
entity='node',
entity_id='',
)
return jsonify({'success': True, 'event_type': 'system_restore_completed'}), 200
except Exception as e:
return jsonify({'error': 'internal_error', 'detail': str(e)}), 500
+19 -15
View File
@@ -40,6 +40,12 @@ TOOL_METADATA = {
'vfio_iommu': {'name': 'VFIO/IOMMU Passthrough', 'function': 'enable_vfio_iommu', 'version': '1.0'},
'lvm_repair': {'name': 'LVM PV Headers Repair', 'function': 'repair_lvm_headers', 'version': '1.0'},
'repo_cleanup': {'name': 'Repository Cleanup', 'function': 'cleanup_repos', 'version': '1.0'},
# 1.1 = setup_proxmox_repositories now re-applies chmod 0644 to existing
# .sources/.list files. Legacy users (repos created with old 0640 perms
# but no entry in installed_tools.json) are surfaced as v1.0 via the
# legacy detector in get_installed_tools() below, so the Settings page
# shows them an "Update available" they can apply without touching apt.
'proxmox_repos': {'name': 'Proxmox APT Repositories', 'function': 'setup_proxmox_repositories', 'version': '1.1'},
# ── Legacy / Deprecated entries ──
# These optimizations were applied by previous ProxMenux versions but are
# no longer needed or have been removed from the current scripts. We still
@@ -226,8 +232,15 @@ def get_installed_tools():
'message': 'No ProxMenux optimizations installed yet'
})
with open(installed_tools_path, 'r') as f:
raw = json.load(f)
# Use the shared loader so both update detection and this
# endpoint see the same set of entries — including any
# synthetic v1.0 entries injected by `_apply_legacy_detectors`
# for tools that the host has configured but were never
# recorded in installed_tools.json (e.g. `proxmox_repos` on a
# pre-1.2.2 install). Without this, the update detector would
# surface a fix as available but the Settings endpoint
# wouldn't list the row, leaving the user nothing to click.
loaded = post_install_versions.load_installed_tools()
# Sprint 12A: index update list by tool key for has_update lookup.
try:
@@ -237,20 +250,11 @@ def get_installed_tools():
update_by_key = {u['key']: u for u in piv_snapshot.get('updates', [])}
tools = []
for tool_key, value in raw.items():
# Normalize legacy bool vs new structured entry.
if isinstance(value, bool):
if not value:
continue
installed_version = '1.0'
source = ''
elif isinstance(value, dict):
if not value.get('installed', False):
continue
installed_version = str(value.get('version', '1.0')) or '1.0'
source = str(value.get('source', '') or '')
else:
for tool_key, value in loaded.items():
if not value.get('installed', False):
continue
installed_version = str(value.get('version', '1.0')) or '1.0'
source = str(value.get('source', '') or '')
# Hard-coded display metadata (display name, deprecated flag).
meta = TOOL_METADATA.get(tool_key, {})
File diff suppressed because it is too large Load Diff
@@ -262,6 +262,10 @@ def terminal_websocket(ws):
_term_env.setdefault('COLORTERM', 'truecolor')
_term_env.setdefault('LANG', 'C.UTF-8')
_term_env.setdefault('LC_ALL', 'C.UTF-8')
# Inherited by every child of this shell (including `menu`), so the
# update path can tell it's running inside a WebSocket-backed session
# that would be cut mid-install if the Monitor service restarted.
_term_env['PROXMENUX_TERMINAL'] = 'monitor'
_term_env.pop('PS1', None)
_home = _term_env.get('HOME') or os.path.expanduser('~') or '/root'
+402 -88
View File
@@ -58,6 +58,157 @@ def _perf_log(section: str, elapsed_ms: float):
if DEBUG_PERF:
print(f"[PERF] {section} = {elapsed_ms:.1f}ms")
# Cap notification `reason` strings so a spike (e.g. 40 CTs above 85%
# rootfs) can't blow past Telegram's 4096-byte cap or truncate the
# subject line in email clients. First 8 names + a tail count keeps
# the message actionable without listing every offender.
_NAME_LIST_LIMIT = 8
def _fmt_name_list(items, limit: int = _NAME_LIST_LIMIT, sep: str = ", ") -> str:
"""Join up to `limit` items into a comma-separated list; when the
input is longer, append " …and N more" so the reader sees both a
concrete sample AND that the list is truncated.
items: iterable of strings.
"""
seq = list(items)
if len(seq) <= limit:
return sep.join(seq)
head = sep.join(seq[:limit])
return f"{head} …and {len(seq) - limit} more"
def _fmt_entity_and_summary(items, singular: str, plural: str, limit: int = _NAME_LIST_LIMIT):
"""Return a (title_entity, reason_summary) tuple for a list of
affected entities. Used by health checks that want to surface both
a headline entity in the notification title AND a full summary in
the body.
- 1 item ("Foo", "Foo <singular>")
- 2 items ("Foo +1", "2 <plural>: Foo, Bar")
- N items ("Foo +N-1", "N <plural>: Foo, Bar, … …and M more")
"""
seq = list(items)
n = len(seq)
if n == 0:
return "", ""
if n == 1:
return seq[0], f"{seq[0]} {singular}"
title_entity = f"{seq[0]} +{n - 1}"
reason = f"{n} {plural}: {_fmt_name_list(seq, limit)}"
return title_entity, reason
# USB-NVMe bridges (ASMedia, JMicron, Realtek) answer plain smartctl with
# the *bridge* identity — model shows as "ASMT 2462 NVME" and there is no
# temperature. Only `-d snt*` passes through to the actual NVMe controller
# behind the bridge. For removable disks we try the snt* variants first
# so both identity and health reflect the drive, not the enclosure.
_USB_NVME_DRIVERS = ('sntasmedia', 'sntjmicron', 'sntrealtek')
def _disk_base_for_sysfs(name: str) -> str:
"""Normalize `/dev/sda` / `sda` to just `sda` for `/sys/block/<name>` lookups."""
if name.startswith('/dev/'):
return name[5:]
return name
_standby_cache: dict = {}
_STANDBY_CACHE_TTL = 15
def _hdd_in_standby(disk_name: str) -> bool:
"""True if `disk_name` is a spinning disk currently parked.
CHECK POWER MODE probe (`smartctl -n standby`) answered without
spinning the drive up. Rotational, non-NVMe only. Mirrors the helper
in flask_server.py; see issue #232.
"""
base = _disk_base_for_sysfs(disk_name)
if base.startswith('nvme'):
return False
now = time.time()
hit = _standby_cache.get(base)
if hit and now - hit[0] < _STANDBY_CACHE_TTL:
return hit[1]
try:
with open(f'/sys/block/{base}/queue/rotational') as f:
if f.read().strip() != '1':
_standby_cache[base] = (now, False)
return False
except OSError:
return False
parked = False
try:
r = subprocess.run(
['smartctl', '-n', 'standby', '-i', f'/dev/{base}'],
capture_output=True, text=True, timeout=5)
parked = r.returncode == 2
except Exception:
parked = False
_standby_cache[base] = (now, parked)
return parked
def _lvm_device_args() -> list:
"""`--devices` list scoping pvs/lvs/vgs off parked disks so they
aren't spun up by the LVM scan (issue #232). Empty when nothing is
parked, so the plain command runs unchanged. Each partition is
listed too a whole-disk path alone doesn't cover a PV on a
partition."""
try:
r = subprocess.run(
['lsblk', '-rno', 'NAME,TYPE,PKNAME'],
capture_output=True, text=True, timeout=5)
if r.returncode != 0:
return []
nodes = []
for line in r.stdout.strip().split('\n'):
parts = line.split()
if len(parts) < 2 or parts[1] not in ('disk', 'part'):
continue
name = parts[0]
base = name if parts[1] == 'disk' else (parts[2] if len(parts) > 2 else name)
nodes.append((name, base))
standby_bases = {
name for name, base in nodes
if name == base and _hdd_in_standby(name)
}
if not standby_bases:
return []
devices = [f'/dev/{name}' for name, base in nodes if base not in standby_bases]
return ['--devices', ','.join(devices)] if devices else []
except Exception:
return []
def _is_disk_removable(disk_name: str) -> bool:
"""True if `/sys/block/<disk>/removable` reads 1. USB-attached storage."""
try:
base = _disk_base_for_sysfs(disk_name)
with open(f'/sys/block/{base}/removable') as f:
return f.read().strip() == '1'
except Exception:
return False
def _is_disk_usb(disk_name: str) -> bool:
"""True if the disk sits behind a USB bus. Reads the resolved sysfs
device path reliable for USB-NVMe bridges and USB-attached HDDs
that report `removable=0` even though they ARE USB (so the older
`_is_disk_removable` heuristic skipped snt* driver probes and left
NVMe-behind-a-bridge disks with the bridge's own chatter cached
forever)."""
try:
base = _disk_base_for_sysfs(disk_name)
real = os.path.realpath(f'/sys/block/{base}')
return any(seg.startswith('usb') and (len(seg) == 3 or seg[3:].isdigit())
for seg in real.split('/'))
except Exception:
return False
class HealthMonitor:
"""
Monitors system health across multiple components with minimal impact.
@@ -1107,16 +1258,16 @@ class HealthMonitor:
# Severity: CRITICAL > WARNING > UNKNOWN (capped at WARNING) > INFO > OK
if critical_issues:
overall = 'CRITICAL'
summary = '; '.join(critical_issues[:3])
summary = _fmt_name_list(critical_issues, sep='; ')
elif warning_issues:
overall = 'WARNING'
summary = '; '.join(warning_issues[:3])
summary = _fmt_name_list(warning_issues, sep='; ')
elif unknown_issues:
overall = 'WARNING' # UNKNOWN caps at WARNING, never escalates to CRITICAL
summary = '; '.join(unknown_issues[:3])
summary = _fmt_name_list(unknown_issues, sep='; ')
elif info_issues:
overall = 'OK' # INFO statuses don't degrade overall health
summary = '; '.join(info_issues[:3])
summary = _fmt_name_list(info_issues, sep='; ')
else:
overall = 'OK'
summary = 'All systems operational'
@@ -1143,15 +1294,24 @@ class HealthMonitor:
cat_status = cat_data.get('status', 'OK')
prev_status = previous_details.get(cat_key, 'OK')
if prev_status != cat_status and cat_status in ('WARNING', 'CRITICAL'):
# `entity` gives the notification title a concrete
# subject (e.g. "Tuxis (dir)") instead of the bare
# category name. Checks that surface a single
# affected object populate it in _check_XXX; the
# rest just fall through and the template renders
# without the placeholder thanks to _SafeDict.
event_data = {
'previous': prev_status,
'current': cat_status,
'reason': cat_data.get('reason', ''),
}
if cat_data.get('entity'):
event_data['entity'] = cat_data['entity']
health_persistence.emit_event(
event_type='state_change',
category=cat_key,
severity=cat_status,
data={
'previous': prev_status,
'current': cat_status,
'reason': cat_data.get('reason', '')
}
data=event_data,
)
self._last_overall_status = overall
@@ -1219,6 +1379,16 @@ class HealthMonitor:
WARNING_MIN_SAMPLES = 25 # ~250s of sustained elevated CPU
RECOVERY_MIN_SAMPLES = 10 # ~100s of recovery
# Build the `details` payload the cpu_high notification
# template (notification_templates.py) expects: `value`
# (already-formatted percent), `cores` (CPU count) and a
# human-readable `details` line. The previous payload only
# carried `cpu_percent`/`duration`, which left the template
# placeholders blank — the user got "High CPU usage — %"
# and "CPU usage has reached % on cores." with no values.
cpu_count = os.cpu_count() or 1
value_str = f"{cpu_percent:.0f}"
if len(critical_samples) >= CRITICAL_MIN_SAMPLES:
# Calculate actual duration from oldest to newest sample
oldest = min(s['time'] for s in critical_samples)
@@ -1231,7 +1401,13 @@ class HealthMonitor:
category='cpu',
severity='CRITICAL',
reason=reason,
details={'cpu_percent': cpu_percent, 'duration': actual_duration}
details={
'value': value_str,
'cores': cpu_count,
'details': f'Sustained for {actual_duration}s above {self.CPU_CRITICAL}%.',
'cpu_percent': cpu_percent,
'duration': actual_duration,
},
)
elif len(warning_samples) >= WARNING_MIN_SAMPLES and len(recovery_samples) < RECOVERY_MIN_SAMPLES:
oldest = min(s['time'] for s in warning_samples)
@@ -1244,7 +1420,13 @@ class HealthMonitor:
category='cpu',
severity='WARNING',
reason=reason,
details={'cpu_percent': cpu_percent, 'duration': actual_duration}
details={
'value': value_str,
'cores': cpu_count,
'details': f'Sustained for {actual_duration}s above {self.CPU_WARNING}%.',
'cpu_percent': cpu_percent,
'duration': actual_duration,
},
)
else:
status = 'OK'
@@ -1283,7 +1465,9 @@ class HealthMonitor:
t_status = temp_status.get('status', 'OK')
checks['cpu_temperature'] = {
'status': t_status,
'detail': 'Temperature elevated' if t_status != 'OK' else 'Normal'
'detail': 'Temperature elevated' if t_status != 'OK' else 'Normal',
'error_key': 'cpu_temperature',
'dismissable': True,
}
else:
checks['cpu_temperature'] = {
@@ -1378,7 +1562,7 @@ class HealthMonitor:
category='temperature',
severity='WARNING',
reason=reason,
details={'temperature': max_temp, 'duration': actual_duration, 'dismissable': False}
details={'temperature': max_temp, 'duration': actual_duration, 'dismissable': True}
)
elif len(recovery_samples) >= 3:
# Temperature has been ≤80°C for 30 seconds - clear the error
@@ -1579,6 +1763,17 @@ class HealthMonitor:
error_key = f'disk_space_{mount_point}'
if fs_status['status'] != 'OK':
issues.append(f"{mount_point}: {fs_status['reason']}")
# Carry error_key + dismissable into the check
# dict — the frontend gates the Dismiss dropdown
# on `dismissable === true` strictly, and the
# spread at the consumer (`storage_details →
# checks[key]`) only forwards fields actually
# present on fs_status. Without these two lines,
# /mnt/* fullness rows reach the UI as plain
# CRITICAL/WARNING with no Dismiss button — the
# same gap the operator hit on stable 1.2.2.
fs_status['error_key'] = error_key
fs_status['dismissable'] = True
storage_details[mount_point] = fs_status
# Record persistent error for notifications
usage = psutil.disk_usage(mount_point)
@@ -1596,7 +1791,7 @@ class HealthMonitor:
'mount': mount_point,
'used': str(round(usage.percent, 1)),
'available': avail_str,
'dismissable': False,
'dismissable': True,
}
)
else:
@@ -1610,10 +1805,19 @@ class HealthMonitor:
if zfs_pool_issues:
for pool_name, pool_info in zfs_pool_issues.items():
issues.append(f'{pool_name}: {pool_info["reason"]}')
storage_details[pool_name] = pool_info
# Record error for notification system
# Carry error_key + dismissable BEFORE pushing — same
# rationale as the disk_space path above: the frontend
# gates the Dismiss button on `dismissable === true`
# strictly, and the splat at the consumer only forwards
# what pool_info already has. real_pool gets computed
# twice (here + below) but both paths land on the same
# key.
real_pool = pool_info.get('pool_name', pool_name)
pool_info['error_key'] = f'zfs_pool_{real_pool}'
pool_info['dismissable'] = True
storage_details[pool_name] = pool_info
# Record error for notification system
zfs_error_key = f'zfs_pool_{real_pool}'
zfs_reason = f'ZFS pool {real_pool}: {pool_info["reason"]}'
try:
@@ -1627,7 +1831,7 @@ class HealthMonitor:
'pool_name': real_pool,
'health': pool_info.get('health', ''),
'device': f'zpool:{real_pool}',
'dismissable': False,
'dismissable': True,
}
)
except Exception:
@@ -1999,22 +2203,22 @@ class HealthMonitor:
# persistent WARNING after user has acknowledged.
return {
'status': 'OK',
'reason': '; '.join(issues[:3]),
'reason': _fmt_name_list(issues, sep='; '),
'details': storage_details,
'checks': checks,
'all_dismissed': True,
}
except Exception:
pass
# Determine overall status
has_critical = any(
d.get('status') == 'CRITICAL' for d in storage_details.values()
)
return {
'status': 'CRITICAL' if has_critical else 'WARNING',
'reason': '; '.join(issues[:3]),
'reason': _fmt_name_list(issues, sep='; '),
'details': storage_details,
'checks': checks
}
@@ -2064,8 +2268,10 @@ class HealthMonitor:
if result_which.returncode != 0:
return {'status': 'OK'} # LVM not installed
# `--devices` keeps lvs from scanning parked HDDs (issue #232).
dev_args = _lvm_device_args()
result = subprocess.run(
['lvs', '--noheadings', '--options', 'lv_name,vg_name,lv_attr'],
['lvs', *dev_args, '--noheadings', '--options', 'lv_name,vg_name,lv_attr'],
capture_output=True,
text=True,
timeout=3
@@ -2089,7 +2295,7 @@ class HealthMonitor:
if not volumes:
# Check if any VGs exist to determine if LVM is truly unconfigured or just inactive
vg_result = subprocess.run(
['vgs', '--noheadings', '--options', 'vg_name'],
['vgs', *dev_args, '--noheadings', '--options', 'vg_name'],
capture_output=True,
text=True,
timeout=3
@@ -2361,15 +2567,36 @@ class HealthMonitor:
result = {'serial': '', 'model': '', '_fp': fp}
try:
dev_path = f'/dev/{disk_name}' if not disk_name.startswith('/') else disk_name
proc = subprocess.run(
['smartctl', '-i', '-j', dev_path],
capture_output=True, text=True, timeout=5
)
if proc.returncode in (0, 4):
import json as _json
data = _json.loads(proc.stdout)
result['serial'] = data.get('serial_number', '')
result['model'] = data.get('model_name', '') or data.get('model_family', '')
# USB-attached disks may sit behind an NVMe bridge: try the
# snt* driver variants first so identity reflects the drive
# (Samsung 990 PRO) rather than the enclosure (ASMT 2462 NVME).
# If all snt* fail, fall through to the plain call — that's
# still correct for USB-SATA sticks and non-USB devices.
# USB detection is by sysfs path (`_is_disk_usb`) rather than
# the `removable` flag, since USB-NVMe and USB-HDD both report
# `removable=0` even though they ARE USB.
attempts = []
if _is_disk_usb(disk_name) or _is_disk_removable(disk_name):
for drv in _USB_NVME_DRIVERS:
attempts.append(['smartctl', '-i', '-j', '-d', drv, dev_path])
attempts.append(['smartctl', '-i', '-j', dev_path])
import json as _json
for cmd in attempts:
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=5)
if proc.returncode not in (0, 4):
continue
try:
data = _json.loads(proc.stdout)
except Exception:
continue
serial = data.get('serial_number', '')
model = data.get('model_name', '') or data.get('model_family', '')
if serial or model:
result['serial'] = serial
result['model'] = model
break
except Exception:
pass
@@ -2402,20 +2629,51 @@ class HealthMonitor:
try:
dev_path = f'/dev/{disk_name}' if not disk_name.startswith('/') else disk_name
result = subprocess.run(
['smartctl', '--health', '-j', dev_path],
capture_output=True, text=True, timeout=5
)
# `-n standby` skips the command (exit code 2, no disk I/O)
# when the drive is parked, preventing the health poller
# from spinning up HDDs that hdparm / hd-idle just put to
# sleep — issue #232. The "UNKNOWN" branch below correctly
# keeps the previous cached result alive on exit code 2.
#
# USB-attached disks may sit behind an NVMe bridge: try snt*
# drivers first so health reflects the actual NVMe controller.
# A bridge that fakes "PASSED" while the drive behind it is
# failing is exactly the false-negative we want to avoid.
# USB detection uses the sysfs path so USB-NVMe bridges (which
# report removable=0) are caught too.
attempts = []
if _is_disk_usb(disk_name) or _is_disk_removable(disk_name):
for drv in _USB_NVME_DRIVERS:
attempts.append(['smartctl', '-n', 'standby', '--health', '-j', '-d', drv, dev_path])
attempts.append(['smartctl', '-n', 'standby', '--health', '-j', dev_path])
import json as _json
data = _json.loads(result.stdout)
passed = data.get('smart_status', {}).get('passed', None)
if passed is True:
smart_result = 'PASSED'
elif passed is False:
smart_result = 'FAILED'
else:
smart_result = None
for cmd in attempts:
result = subprocess.run(cmd, capture_output=True, text=True, timeout=5)
if result.returncode == 2:
# Drive in standby — reuse the previous health state
# if we have one, otherwise report UNKNOWN. Either way,
# don't refresh the cache TTL so we retry on the next
# cycle (a drive can come out of standby at any time).
if cached:
return cached['result']
return 'UNKNOWN'
try:
data = _json.loads(result.stdout)
except Exception:
continue
passed = data.get('smart_status', {}).get('passed', None)
if passed is True:
smart_result = 'PASSED'
break
if passed is False:
smart_result = 'FAILED'
break
# No opinion yet — next attempt (fallthrough to plain).
if smart_result is None:
smart_result = 'UNKNOWN'
# Cache the result with the device fingerprint for hot-swap invalidation
self._smart_cache[cache_key] = {'result': smart_result, 'time': current_time, 'fp': fp}
return smart_result
@@ -2725,7 +2983,7 @@ class HealthMonitor:
'device': disk,
'error_count': error_count,
'smart_status': smart_health,
'dismissable': False,
'dismissable': True,
'error_key': error_key,
}
elif smart_ok:
@@ -2744,7 +3002,7 @@ class HealthMonitor:
'device': disk,
'error_count': error_count,
'smart_status': smart_health,
'dismissable': False,
'dismissable': True,
'error_key': error_key,
}
elif smart_health == 'FAILED':
@@ -2764,7 +3022,7 @@ class HealthMonitor:
'serial': resolved_serial or '',
'error_count': error_count,
'smart_status': smart_health,
'sample': sample, 'dismissable': False}
'sample': sample, 'dismissable': True}
)
disk_results[display] = {
'status': severity,
@@ -2772,7 +3030,7 @@ class HealthMonitor:
'device': disk,
'error_count': error_count,
'smart_status': smart_health,
'dismissable': False,
'dismissable': True,
'error_key': error_key,
}
else:
@@ -2894,10 +3152,17 @@ class HealthMonitor:
}
has_critical = any(d.get('status') == 'CRITICAL' for d in active_results.values())
# Surface disk names in the reason so notifications identify
# which device(s) took the hit rather than just a count.
disk_names = sorted(active_results.keys())
entity, summary = _fmt_entity_and_summary(
disk_names, singular='has errors', plural='disks with errors',
)
return {
'status': 'CRITICAL' if has_critical else 'WARNING',
'reason': f"{len(active_results)} disk(s) with errors",
'reason': summary,
'entity': entity,
'details': disk_results
}
@@ -2991,13 +3256,14 @@ class HealthMonitor:
category='network',
severity='CRITICAL',
reason=alert_reason or 'Interface DOWN',
details={'interface': interface, 'dismissable': False}
details={'interface': interface, 'dismissable': True}
)
interface_details[interface] = {
'status': 'CRITICAL',
'reason': alert_reason or 'Interface DOWN',
'dismissable': False
'dismissable': True,
'error_key': error_key,
}
else:
active_interfaces.add(interface)
@@ -3028,7 +3294,8 @@ class HealthMonitor:
checks[iface] = {
'status': detail.get('status', 'OK'),
'detail': detail.get('reason', 'DOWN'),
'dismissable': detail.get('dismissable', False)
'dismissable': detail.get('dismissable', True),
'error_key': detail.get('error_key') or iface,
}
checks['connectivity'] = connectivity_check
@@ -3039,11 +3306,11 @@ class HealthMonitor:
return {
'status': 'CRITICAL' if has_critical else 'WARNING',
'reason': '; '.join(issues[:2]),
'reason': _fmt_name_list(issues, sep='; '),
'details': interface_details,
'checks': checks
}
except Exception as e:
print(f"[HealthMonitor] Network check failed: {e}")
return {'status': 'UNKNOWN', 'reason': f'Network check unavailable: {str(e)}', 'checks': {}, 'dismissable': True}
@@ -3383,10 +3650,10 @@ class HealthMonitor:
return {
'status': 'CRITICAL' if has_critical else 'WARNING',
'reason': '; '.join(issues[:3]),
'reason': _fmt_name_list(issues, sep='; '),
'details': vm_details
}
except Exception as e:
print(f"[HealthMonitor] VMs/CTs check failed: {e}")
return {'status': 'UNKNOWN', 'reason': f'VM/CT check unavailable: {str(e)}', 'checks': {}, 'dismissable': True}
@@ -3418,26 +3685,32 @@ class HealthMonitor:
if health_persistence.check_vm_running(vm_id):
continue # Error auto-resolved if VM is now running
# Still active, add to details
# Still active, add to details. `details` may be persisted
# as SQL NULL / JSON null → deserializes to Python None, and
# `dict.get('details', {})` returns None (not `{}`) in that
# case. Coalesce explicitly to avoid `NoneType has no
# attribute 'get'` (issue #255 in 1.2.3).
details = error.get('details') or {}
vm_details[error_key] = {
'status': error['severity'],
'reason': error['reason'],
'id': error.get('details', {}).get('id', 'unknown'),
'type': error.get('details', {}).get('type', 'VM/CT'),
'id': details.get('id', 'unknown'),
'type': details.get('type', 'VM/CT'),
'first_seen': error['first_seen'],
'dismissed': False,
}
issues.append(f"{error.get('details', {}).get('type', 'VM')} {error.get('details', {}).get('id', '')}: {error['reason']}")
issues.append(f"{details.get('type', 'VM')} {details.get('id', '')}: {error['reason']}")
# Process dismissed errors (show as INFO)
for error in dismissed_vm_errors:
error_key = error['error_key']
if error_key not in vm_details: # Don't overwrite active errors
details = error.get('details') or {}
vm_details[error_key] = {
'status': 'INFO',
'reason': error['reason'],
'id': error.get('details', {}).get('id', 'unknown'),
'type': error.get('details', {}).get('type', 'VM/CT'),
'id': details.get('id', 'unknown'),
'type': details.get('type', 'VM/CT'),
'first_seen': error['first_seen'],
'dismissed': True,
}
@@ -3613,11 +3886,11 @@ class HealthMonitor:
return {
'status': 'CRITICAL' if has_critical else 'WARNING',
'reason': '; '.join(issues[:3]),
'reason': _fmt_name_list(issues, sep='; '),
'details': vm_details,
'checks': checks
}
except Exception as e:
print(f"[HealthMonitor] VMs/CTs persistence check failed: {e}")
return {'status': 'UNKNOWN', 'reason': f'VM/CT check unavailable: {str(e)}', 'checks': {}, 'dismissable': True}
@@ -4228,7 +4501,12 @@ class HealthMonitor:
'log_persistent_errors': {'active': persistent_count > 0, 'severity': 'WARNING',
'reason': f'{persistent_count} recurring pattern(s) over 15+ min:\n' + '\n'.join(f' - {s}' for s in persist_samples) if persistent_count else ''},
'log_critical_errors': {'active': unique_critical_count > 0, 'severity': 'CRITICAL',
'reason': f'{unique_critical_count} critical error(s) found', 'dismissable': False},
'reason': (
self._enrich_critical_log_reason(next(iter(critical_errors_found.values())))
if unique_critical_count == 1
else f'{unique_critical_count} critical errors:\n' + self._enrich_critical_log_reason(next(iter(critical_errors_found.values())))
) if unique_critical_count > 0 else '',
'dismissable': True},
}
# Track which sub-checks were dismissed
@@ -4298,7 +4576,7 @@ class HealthMonitor:
'log_critical_errors': {
'status': _log_check_status('log_critical_errors', unique_critical_count > 0, 'CRITICAL'),
'detail': reason if unique_critical_count > 0 else 'No critical errors',
'dismissable': False,
'dismissable': True,
'error_key': 'log_critical_errors'
}
}
@@ -4318,7 +4596,7 @@ class HealthMonitor:
detail = v.get('detail', '')
if detail:
active_reasons.append(detail)
reason = '; '.join(active_reasons[:3]) if active_reasons else None
reason = _fmt_name_list(active_reasons, sep='; ') if active_reasons else None
log_result = {'status': status, 'checks': log_checks}
if reason:
@@ -4623,7 +4901,7 @@ class HealthMonitor:
health_persistence.record_error(
error_key='system_age', category='updates',
severity='CRITICAL', reason=reason,
details={'days': last_update_days, 'update_count': update_count, 'dismissable': False}
details={'days': last_update_days, 'update_count': update_count, 'dismissable': True}
)
elif last_update_days and last_update_days >= 365:
status = 'WARNING'
@@ -5071,7 +5349,7 @@ class HealthMonitor:
overall_status = 'CRITICAL' if has_critical else 'WARNING'
return {
'status': overall_status,
'reason': '; '.join(active_issues[:2]),
'reason': _fmt_name_list(active_issues, sep='; '),
'checks': checks
}
@@ -5557,7 +5835,7 @@ class HealthMonitor:
'storage_name': storage_name,
'storage_type': storage.get('type', 'unknown'),
'status_detail': status_detail,
'dismissable': False
'dismissable': True
}
)
@@ -5566,7 +5844,7 @@ class HealthMonitor:
'reason': reason,
'type': storage.get('type', 'unknown'),
'status': status_detail,
'dismissable': False
'dismissable': True
}
# Build checks from storage_details
@@ -5577,14 +5855,14 @@ class HealthMonitor:
checks[st_name] = {
'status': 'INFO',
'detail': f"[Startup] {st_info.get('reason', 'Unavailable')} (checking...)",
'dismissable': False,
'dismissable': True,
'grace_period': True
}
else:
checks[st_name] = {
'status': 'CRITICAL',
'detail': st_info.get('reason', 'Unavailable'),
'dismissable': False
'dismissable': True
}
# Add excluded unavailable storages as INFO (not as errors)
@@ -5609,11 +5887,22 @@ class HealthMonitor:
# Determine overall status based on non-excluded issues only
if real_unavailable:
# Build a name list like "Tuxis (dir), backups-nfs (nfs)"
# so both the notification body and the (headline) title
# can tell the user which storages actually went down —
# the previous "N storage(s) unavailable" wording forced
# them to open the UI to find out.
labels = [f"{s['name']} ({s.get('type', 'unknown')})" for s in real_unavailable]
entity, summary = _fmt_entity_and_summary(
labels, singular='unavailable', plural='Proxmox storages unavailable',
)
# During grace period, return INFO instead of CRITICAL
if in_grace_period:
grace_summary = f"{summary} (startup)" if len(labels) > 1 else f"{labels[0]} not yet available (startup)"
return {
'status': 'INFO',
'reason': f'{len(real_unavailable)} storage(s) not yet available (startup)',
'reason': grace_summary,
'entity': entity,
'details': storage_details,
'checks': checks,
'grace_period': True
@@ -5621,7 +5910,8 @@ class HealthMonitor:
else:
return {
'status': 'CRITICAL',
'reason': f'{len(real_unavailable)} Proxmox storage(s) unavailable',
'reason': summary,
'entity': entity,
'details': storage_details,
'checks': checks
}
@@ -5802,15 +6092,23 @@ class HealthMonitor:
health_persistence.clear_error(ek)
if critical_targets:
entity, summary = _fmt_entity_and_summary(
critical_targets, singular='is stale', plural='remote mounts stale',
)
return {
'status': 'CRITICAL',
'reason': f'{len(critical_targets)} remote mount(s) stale: {", ".join(critical_targets[:3])}',
'reason': summary,
'entity': entity,
'checks': checks,
}
if warning_targets:
entity, summary = _fmt_entity_and_summary(
warning_targets, singular='is read-only', plural='remote mounts read-only',
)
return {
'status': 'WARNING',
'reason': f'{len(warning_targets)} remote mount(s) read-only: {", ".join(warning_targets[:3])}',
'reason': summary,
'entity': entity,
'checks': checks,
}
return {
@@ -5943,15 +6241,19 @@ class HealthMonitor:
if not checks:
return None
if critical_cts:
entity, _ = _fmt_entity_and_summary(critical_cts, 'x', 'x')
return {
'status': 'CRITICAL',
'reason': f'{len(critical_cts)} CT(s) at >{CRIT_PCT}% rootfs: {", ".join(critical_cts[:3])}',
'reason': f'{len(critical_cts)} CT(s) at >{CRIT_PCT}% rootfs: {_fmt_name_list(critical_cts)}',
'entity': entity,
'checks': checks,
}
if warning_cts:
entity, _ = _fmt_entity_and_summary(warning_cts, 'x', 'x')
return {
'status': 'WARNING',
'reason': f'{len(warning_cts)} CT(s) at >{WARN_PCT}% rootfs: {", ".join(warning_cts[:3])}',
'reason': f'{len(warning_cts)} CT(s) at >{WARN_PCT}% rootfs: {_fmt_name_list(warning_cts)}',
'entity': entity,
'checks': checks,
}
return {
@@ -6065,15 +6367,19 @@ class HealthMonitor:
health_persistence.clear_error(ek)
if critical_labels:
entity, _ = _fmt_entity_and_summary(critical_labels, 'x', 'x')
return {
'status': 'CRITICAL',
'reason': f'{len(critical_labels)} CT mount(s) ≥{crit_pct:.0f}%: {", ".join(critical_labels[:3])}',
'reason': f'{len(critical_labels)} CT mount(s) ≥{crit_pct:.0f}%: {_fmt_name_list(critical_labels)}',
'entity': entity,
'checks': checks,
}
if warning_labels:
entity, _ = _fmt_entity_and_summary(warning_labels, 'x', 'x')
return {
'status': 'WARNING',
'reason': f'{len(warning_labels)} CT mount(s) ≥{warn_pct:.0f}%: {", ".join(warning_labels[:3])}',
'entity': entity,
'reason': f'{len(warning_labels)} CT mount(s) ≥{warn_pct:.0f}%: {_fmt_name_list(warning_labels)}',
'checks': checks,
}
return {
@@ -6184,15 +6490,19 @@ class HealthMonitor:
if not checks:
return None # No block storages on this host
if critical_labels:
entity, _ = _fmt_entity_and_summary(critical_labels, 'x', 'x')
return {
'status': 'CRITICAL',
'reason': f'{len(critical_labels)} PVE storage(s) ≥{crit_pct:.0f}%: {", ".join(critical_labels[:3])}',
'reason': f'{len(critical_labels)} PVE storage(s) ≥{crit_pct:.0f}%: {_fmt_name_list(critical_labels)}',
'entity': entity,
'checks': checks,
}
if warning_labels:
entity, _ = _fmt_entity_and_summary(warning_labels, 'x', 'x')
return {
'status': 'WARNING',
'reason': f'{len(warning_labels)} PVE storage(s) ≥{warn_pct:.0f}%: {", ".join(warning_labels[:3])}',
'reason': f'{len(warning_labels)} PVE storage(s) ≥{warn_pct:.0f}%: {_fmt_name_list(warning_labels)}',
'entity': entity,
'checks': checks,
}
return {
@@ -6292,15 +6602,19 @@ class HealthMonitor:
health_persistence.clear_error(ek)
if critical_labels:
entity, _ = _fmt_entity_and_summary(critical_labels, 'x', 'x')
return {
'status': 'CRITICAL',
'reason': f'{len(critical_labels)} ZFS pool(s) ≥{crit_pct:.0f}%: {", ".join(critical_labels[:3])}',
'reason': f'{len(critical_labels)} ZFS pool(s) ≥{crit_pct:.0f}%: {_fmt_name_list(critical_labels)}',
'entity': entity,
'checks': checks,
}
if warning_labels:
entity, _ = _fmt_entity_and_summary(warning_labels, 'x', 'x')
return {
'status': 'WARNING',
'reason': f'{len(warning_labels)} ZFS pool(s) ≥{warn_pct:.0f}%: {", ".join(warning_labels[:3])}',
'reason': f'{len(warning_labels)} ZFS pool(s) ≥{warn_pct:.0f}%: {_fmt_name_list(warning_labels)}',
'entity': entity,
'checks': checks,
}
return {
+91 -5
View File
@@ -487,7 +487,55 @@ class HealthPersistence:
conn.close()
def record_error(self, error_key: str, category: str, severity: str,
@staticmethod
def _entity_from_details(details: Optional[Dict]) -> str:
"""Derive a short display identifier for the affected object from
the details blob so notification titles can name it. Returns ''
when no per-object identity is available (aggregate-only checks).
Order of preference matches what users actually recognize:
friendly name first, then id-with-name, then bare id, then path.
"""
if not details:
return ''
d = details
# Storage / mounts / pools
if d.get('storage_name'):
st = d.get('storage_type') or d.get('type')
return f"{d['storage_name']} ({st})" if st else d['storage_name']
if d.get('pool_name'):
return d['pool_name']
if d.get('mount_point'):
return d['mount_point']
if d.get('mount_target'):
return d['mount_target']
# VMs / CTs
if d.get('vm_name') and d.get('vmid'):
return f"{d['vm_name']} ({d['vmid']})"
if d.get('ct_name') and d.get('vmid'):
return f"{d['ct_name']} ({d['vmid']})"
if d.get('vmid'):
return str(d['vmid'])
# Disks / devices
if d.get('device'):
return d['device']
if d.get('disk'):
return d['disk']
# Network / services
if d.get('interface'):
return d['interface']
if d.get('iface'):
return d['iface']
if d.get('service_name'):
return d['service_name']
if d.get('service'):
return d['service']
# Generic fallback
if d.get('name'):
return str(d['name'])
return ''
def record_error(self, error_key: str, category: str, severity: str,
reason: str, details: Optional[Dict] = None) -> Dict[str, Any]:
"""
Record or update an error.
@@ -601,6 +649,8 @@ class HealthPersistence:
event_info = {'type': 'new', 'needs_notification': True}
self._record_event(cursor, 'new', error_key,
{'severity': severity, 'reason': reason,
'entity': self._entity_from_details(details),
'details': details,
'note': 'Re-triggered after suppression expired'})
conn.commit()
return event_info
@@ -653,9 +703,15 @@ class HealthPersistence:
conn.commit()
return event_info
# Record event
# Record event with the caller-supplied `details` so downstream
# notification templates can render the affected object's name
# (storage_name, vm_name, mount_point, …) in the title/body
# via `_SafeDict`. Without this, only the aggregate `reason`
# travels and titles fall back to bare category labels.
self._record_event(cursor, event_info['type'], error_key,
{'severity': severity, 'reason': reason})
{'severity': severity, 'reason': reason,
'entity': self._entity_from_details(details),
'details': details})
conn.commit()
finally:
@@ -690,7 +746,26 @@ class HealthPersistence:
''', (now, reason, error_key))
if cursor.rowcount > 0:
self._record_event(cursor, 'resolved', error_key, {'reason': reason})
# Reload the resolved error's details so the resolution
# event can name the same entity that was named when it
# was created — otherwise "Storage 'Tuxis' unavailable"
# comes back as "Resolved - Storage" with no identity.
cursor.execute(
'SELECT details FROM errors WHERE error_key = ? ORDER BY id DESC LIMIT 1',
(error_key,),
)
row = cursor.fetchone()
stored_details = None
if row and row[0]:
try:
stored_details = json.loads(row[0])
except Exception:
stored_details = None
self._record_event(cursor, 'resolved', error_key, {
'reason': reason,
'entity': self._entity_from_details(stored_details),
'details': stored_details or {},
})
conn.commit()
@@ -843,11 +918,22 @@ class HealthPersistence:
# Try to infer category from the error_key prefix.
category = ''
# Order matters: more specific prefixes MUST come before shorter ones
# e.g. 'security_updates' (updates) before 'security_' (security)
# e.g. 'security_updates' (updates) before 'security_' (security),
# and 'lxc_disk_low_' / 'zfs_pool_full_' (storage) before the shorter
# 'disk_' / 'zfs_pool_' fallbacks that map to 'disks'.
for cat, prefix in [('updates', 'security_updates'), ('updates', 'system_age'),
('updates', 'pending_updates'), ('updates', 'kernel_pve'),
('security', 'security_'),
('pve_services', 'pve_service_'), ('vms', 'vmct_'), ('vms', 'vm_'), ('vms', 'ct_'),
# ── Storage keys — HealthMonitor emits these under `storage` category
# but they used to fall through to 'general' here because no prefix
# matched, breaking the Dismiss flow (the acknowledge would persist
# but the storage cache wouldn't be invalidated because the ack was
# tagged with the wrong category).
('storage', 'storage_unavailable_'), ('storage', 'mount_stale'),
('storage', 'mount_readonly'), ('storage', 'lxc_disk_low_'),
('storage', 'lxc_mount_low_'), ('storage', 'pve_storage_full_'),
('storage', 'zfs_pool_full_'),
('disks', 'disk_smart_'), ('disks', 'disk_'), ('disks', 'smart_'), ('disks', 'zfs_pool_'),
('logs', 'log_'), ('network', 'net_'),
('temperature', 'temp_')]:
+41
View File
@@ -66,6 +66,47 @@ def require_auth(f):
return decorated_function
def require_auth_or_ticket(f):
"""Like `require_auth` but ALSO accepts a single-use `?ticket=...`
query parameter (same tickets `/api/terminal/ticket` issues for
WebSockets). Use on endpoints that the browser needs to invoke
from an <a download href="..."> tag anchor tags can't send the
Authorization header, so the caller fetches a ticket first and
appends it to the URL.
Use only for streaming downloads where adding `?ticket=...` is
the only practical way to authenticate; everything else stays
on plain Bearer-token auth."""
@wraps(f)
def decorated_function(*args, **kwargs):
config = load_auth_config()
if not config.get("enabled", False) or config.get("declined", False):
return f(*args, **kwargs)
# First try the Bearer header (fetch from JS uses this).
auth_header = request.headers.get('Authorization')
if auth_header:
parts = auth_header.split()
if len(parts) == 2 and parts[0].lower() == 'bearer':
if verify_token(parts[1]):
return f(*args, **kwargs)
# Fall through to single-use ticket (works for <a download>).
try:
from flask_terminal_routes import _consume_terminal_ticket
if _consume_terminal_ticket(request.args.get('ticket', '')):
return f(*args, **kwargs)
except ImportError:
pass
return jsonify({
"error": "Authentication required",
"message": "Provide a Bearer token in the Authorization header or a fresh ?ticket=... from /api/terminal/ticket"
}), 401
return decorated_function
def require_admin_scope(f):
"""Like `require_auth` but ALSO requires the token's `scope == full_admin`.
+9 -1
View File
@@ -983,6 +983,10 @@ class EmailChannel(NotificationChannel):
elif group == 'backup':
_add('VM/CT ID', data.get('vmid'), 'code')
_add('Name', data.get('vmname'), 'bold')
# Storage / destination — the piece a multi-PBS operator needs to
# tell which target the backup ran against. Reported gap: emails
# showed no way to distinguish which PBS failed with 2+ configured.
_add('Storage', data.get('storage') or data.get('storage_name'), 'code')
_add('Status', 'Failed' if 'fail' in event_type else 'Completed' if 'complete' in event_type else 'Started',
'severity' if 'fail' in event_type else '')
_add('Size', data.get('size'))
@@ -1082,7 +1086,11 @@ class EmailChannel(NotificationChannel):
)
rows.append((esc('Important Packages'), pkg_html))
_add('Current Version', data.get('current_version'), 'code')
_add('New Version', data.get('new_version'), 'code')
# `new_version` is the field used by generic package-update events;
# driver-update templates (nvidia, coral) populate `latest_version`.
# Read both so the tabular row is never empty when the template's
# title/body already printed the new version.
_add('New Version', data.get('new_version') or data.get('latest_version'), 'code')
# ── Other / unknown ──
else:
+268 -18
View File
@@ -396,6 +396,74 @@ def is_vzdump_active_on_host() -> bool:
return found
# ─── APT / dpkg activity gate ────────────────────────────────────
# PVE services (pve-cluster, pveproxy, pvedaemon, corosync…) are
# routinely killed and restarted as part of a package upgrade, so
# every full-upgrade produces a burst of `service_fail` events that
# are entirely expected. We gate `_check_service_failure` on this
# helper so those events are silently dropped while apt is running
# (plus a short grace window after it exits).
_APT_ACTIVE_MARKER = '/var/run/proxmenux-update-in-progress'
_APT_FINISHED_MARKER = '/var/run/proxmenux-update-just-finished'
_APT_GRACE_SECONDS = 60
_DPKG_LOCK_FILE = '/var/lib/dpkg/lock-frontend'
_APT_ACTIVE_CACHE_TTL = 3.0
_apt_active_cache_ts = 0.0
_apt_active_cache_value = False
def is_apt_active_on_host() -> bool:
"""Return True while apt/dpkg is running on this host.
Sources checked, in order:
1. `/var/run/proxmenux-update-in-progress` created by
`scripts/utilities/proxmox_update.sh` around its full-upgrade
call so ProxMenux-driven updates are always covered.
2. `fuser` on `/var/lib/dpkg/lock-frontend` covers a manual
`apt`/`dpkg`/`apt-get` invocation by the operator, or any
other tool holding the lock.
3. Grace window (60 s) after `/var/run/proxmenux-update-just-finished`
was touched catches the tail of service_fail events that
only reach the journal shortly after apt itself exits.
Cached 3 s so a burst of journal events doesn't spawn a fuser
subprocess per line. Caller-safe: returns False on any error.
"""
global _apt_active_cache_ts, _apt_active_cache_value
now = time.time()
if now - _apt_active_cache_ts < _APT_ACTIVE_CACHE_TTL:
return _apt_active_cache_value
active = False
try:
if os.path.exists(_APT_ACTIVE_MARKER):
active = True
elif os.path.exists(_APT_FINISHED_MARKER):
try:
if now - os.path.getmtime(_APT_FINISHED_MARKER) < _APT_GRACE_SECONDS:
active = True
except OSError:
pass
if not active:
try:
result = subprocess.run(
['fuser', _DPKG_LOCK_FILE],
capture_output=True, timeout=2,
)
if result.returncode == 0 and result.stdout.strip():
active = True
except (subprocess.TimeoutExpired, FileNotFoundError, OSError):
pass
except Exception:
active = False
_apt_active_cache_ts = now
_apt_active_cache_value = active
return active
# ─── Journal Watcher (Real-time) ─────────────────────────────────
class JournalWatcher:
@@ -597,22 +665,50 @@ class JournalWatcher:
if self._running:
time.sleep(5) # Wait before restart
# Discard a saved cursor older than this many seconds — resuming
# from a stale cursor replays every event journald has produced
# since the file was last touched, which surfaces as duplicate
# notifications. 15 min comfortably covers graceful shutdowns,
# cron-restart loops and brief outages while keeping a multi-day
# downtime from spamming the operator. Reported on RimegraVE
# (.1.10): the service was last stopped 11 days before a restart;
# on respawn the JournalWatcher fired 16 backup_start notifs in
# 40 min for vzdump runs that PVE had logged hours-to-days
# earlier. source=journal in notification_history proved the
# event chain came from this replay path.
_CURSOR_STALE_AFTER_SEC = 15 * 60
def _run_journalctl(self):
"""Run journalctl -f and process output line by line."""
# Persist the cursor across watcher restarts so we don't lose events
# in the 5s gap between subprocess crash and respawn. journalctl
# writes the file with the latest seen cursor and on next start
# resumes from there. Falls back to -n 0 (start from now) only on
# the very first run when the cursor file doesn't exist yet.
"""Run journalctl -f and process output line by line.
The cursor file lets us pick up after a short watcher crash
without dropping events, but we drop it before invoking
journalctl if it's older than _CURSOR_STALE_AFTER_SEC.
That forces journalctl back to `-n 0` (start from now)
instead of replaying days of history as if it were live.
"""
cursor_file = '/usr/local/share/proxmenux/journal_cursor.txt'
try:
Path(cursor_file).parent.mkdir(parents=True, exist_ok=True)
except Exception:
pass
cursor_path = Path(cursor_file)
if cursor_path.exists():
try:
age = time.time() - cursor_path.stat().st_mtime
if age > self._CURSOR_STALE_AFTER_SEC:
cursor_path.unlink(missing_ok=True)
print(
f"[JournalWatcher] Cursor stale "
f"({int(age)}s old) — restarting from now to "
f"avoid replaying historical events"
)
except OSError:
pass
cmd = ['journalctl', '-f', '-o', 'json', '--no-pager',
f'--cursor-file={cursor_file}']
if not Path(cursor_file).exists():
cmd.extend(['-n', '0']) # First run: don't replay history
if not cursor_path.exists():
cmd.extend(['-n', '0']) # First run (or stale): don't replay history
self._process = subprocess.Popen(
cmd, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL,
@@ -773,7 +869,7 @@ class JournalWatcher:
if re.search(pattern, msg, re.IGNORECASE):
entity = 'node'
entity_id = ''
# Build a context-rich reason from the journal message.
enriched = reason
@@ -931,7 +1027,7 @@ class JournalWatcher:
enriched = f"{reason}\n{msg[:300]}"
data = {'reason': enriched, 'hostname': self._hostname}
self._emit(event_type, severity, data, entity=entity, entity_id=entity_id)
return
@@ -1023,6 +1119,14 @@ class JournalWatcher:
def _check_service_failure(self, msg: str, unit: str):
"""Detect critical service failures with enriched context."""
# Skip while apt/dpkg is running. PVE services (pve-cluster,
# pveproxy, pvedaemon, corosync…) get killed and restarted as a
# normal part of every package upgrade, so their `service_fail`
# events during that window are expected noise, not real
# failures. See `is_apt_active_on_host()` for detection details.
if is_apt_active_on_host():
return
# Filter out noise -- these are normal systemd transient units,
# not real service failures worth alerting about.
_NOISE_PATTERNS = [
@@ -1772,7 +1876,51 @@ class TaskWatcher:
except Exception as e:
# Log error for debugging but return status as fallback
return status
def _get_task_context(self, upid: str, task_type: str) -> dict:
"""Extract task-type-specific fields from a task log.
For snapshots and migrations the interesting metadata (snapshot
name, target node) lives inside the task log body not in the
UPID itself. Without it, `snapshot_complete` bodies render as
`Snapshot "" created` and `migration_complete` as `... migrated
to node .` both real rendering bugs.
"""
wanted = None
if task_type in ('qmsnapshot', 'vzsnapshot'):
wanted = 'snapshot'
elif task_type in ('qmigrate', 'vzmigrate'):
wanted = 'migration'
if wanted is None:
return {}
try:
parts = upid.split(':')
if len(parts) < 5:
return {}
starttime_hex = parts[4]
if not starttime_hex:
return {}
subdir = starttime_hex[-1].upper()
log_path = os.path.join(self.TASK_DIR, subdir, upid)
if not os.path.exists(log_path):
return {}
with open(log_path, 'r', errors='replace') as f:
head = ''.join(f.readline() for _ in range(30))
ctx: dict = {}
if wanted == 'snapshot':
m = (re.search(r"snapshot\s+['\"]([^'\"\n]{1,80})['\"]", head, re.IGNORECASE)
or re.search(r"snapshot(?:\s+name)?\s+([A-Za-z0-9._\-]{1,80})", head, re.IGNORECASE))
if m:
ctx['snapshot_name'] = m.group(1).strip()
elif wanted == 'migration':
m = (re.search(r"to\s+node\s+([A-Za-z0-9._\-]{1,60})", head, re.IGNORECASE)
or re.search(r"migration to\s+(?:node\s+)?([A-Za-z0-9._\-]{1,60})", head, re.IGNORECASE))
if m:
ctx['target_node'] = m.group(1).strip()
return ctx
except Exception:
return {}
# Map PVE task types to our event types
TASK_MAP = {
'qmstart': ('vm_start', 'INFO'),
@@ -2083,16 +2231,21 @@ class TaskWatcher:
reason = self._get_task_log_reason(upid, status)
else:
reason = ''
# Populate task-type-specific fields (target_node for migrations,
# snapshot_name for snapshots) so the corresponding template bodies
# don't render "to node ." or `Snapshot ""`.
ctx = self._get_task_context(upid, task_type)
data = {
'vmid': vmid,
'vmname': vmname or f'ID {vmid}',
'hostname': self._hostname,
'user': user,
'reason': reason,
'target_node': '',
'target_node': ctx.get('target_node', ''),
'size': '',
'snapshot_name': '',
'snapshot_name': ctx.get('snapshot_name', ''),
}
# Determine entity type from task type
@@ -2516,12 +2669,17 @@ class PollingCollector:
if not error_key:
continue
# Keep the raw `details` blob in the tracker so recovery
# notifications later can name the same entity ("Storage
# 'Tuxis'") that the original alert did — without this, the
# resolved notification defaults to the bare category.
current_keys[error_key] = {
'category': error.get('category', ''),
'severity': error.get('severity', 'WARNING'),
'reason': error.get('reason', ''),
'first_seen': error.get('first_seen', ''),
'error_key': error_key,
'details': error.get('details'),
}
category = error.get('category', '')
severity = error.get('severity', 'WARNING')
@@ -2852,6 +3010,18 @@ class PollingCollector:
'duration': duration_label,
'is_recovery': True,
}
# Spread the original details blob so the resolved notification
# can use the same {storage_name}/{vm_name}/{device} placeholders
# the alert did. Mirrors what the alert path does on line ~2718.
resolved_details = old_meta.get('details')
if isinstance(resolved_details, str):
try:
resolved_details = json.loads(resolved_details)
except (json.JSONDecodeError, TypeError):
resolved_details = None
if isinstance(resolved_details, dict):
for _k, _v in resolved_details.items():
data.setdefault(_k, _v)
self._queue.put(NotificationEvent(
'error_resolved', 'OK', data, source='health',
@@ -2972,7 +3142,8 @@ class PollingCollector:
'system_startup', severity, data, source='polling',
entity='node', entity_id='',
))
startup_grace.mark_startup_aggregated()
# ── Update check (enriched) ────────────────────────────────
# Proxmox-related package prefixes used for categorisation
@@ -3475,13 +3646,20 @@ class PollingCollector:
about the final shape."""
item_type = item.get('type', '')
update = item.get('update_check', {}) or {}
# Version fallbacks: if `latest` is missing (checker couldn't
# determine an upstream — network hiccup, or the app itself
# isn't in the update list because only sidecar packages need
# updating), anchor to the current version so template
# placeholders never render as "v" with nothing after.
_cur = item.get('current_version') or ''
_lat = update.get('latest') or _cur or 'unknown'
common = {
'hostname': self._hostname,
'name': item.get('name') or item.get('id'),
'menu_label': item.get('menu_label') or '',
'menu_script': item.get('menu_script') or '',
'current_version': item.get('current_version') or '',
'latest_version': update.get('latest') or '',
'current_version': _cur or 'unknown',
'latest_version': _lat,
}
if item_type == 'oci_app':
@@ -3491,12 +3669,24 @@ class PollingCollector:
f"{p.get('latest', '?')}"
for p in packages
]
# Decide the wording based on whether Tailscale itself moved.
# When current == latest, showing "v1.90 → v1.90" reads as a
# bug ("update to same version?"); switch to a neutral line
# that makes it clear only sidecar packages are updating.
if _cur and _lat and _cur == _lat:
update_title_suffix = f' — v{_cur} (packages only)'
version_line = f'🔹 Tailscale: v{_cur} (unchanged — only sidecar packages need updating)'
else:
update_title_suffix = f' — v{_lat}'
version_line = f'🔹 Current Tailscale: v{_cur or "unknown"} → 🟢 Latest: v{_lat}'
data = {
**common,
'app_id': item.get('id', '').removeprefix('oci:'),
'app_name': common['name'],
'package_count': len(packages),
'package_list': '\n'.join(pkg_lines) or ' (no detail)',
'update_title_suffix': update_title_suffix,
'version_line': version_line,
}
return 'secure_gateway_update_available', data
@@ -3827,6 +4017,44 @@ class ProxmoxHookWatcher:
'title': title or event_type,
'job_id': pve_job_id,
}
# `system_problem` is the generic fallback of `_classify_pve` for
# unknown/empty pve_type. Without a populated `reason`, the template
# renders "Reason: " (empty) and `_summarize_event` falls back to
# printing the raw event_type ("problema_del_sistema") in every
# `burst_system` aggregate, producing the useless
# "🔵 constructor: Problema del sistema detectado" / "+1 problema
# más del sistema (Problemas adicionales: - problema_del_sistema)"
# messages the operator sees on Telegram. Preserve the PVE payload
# as `reason` so both surfaces have concrete text to render.
if event_type == 'system_problem':
reason_text = (message or title or '').strip()
if reason_text:
data['reason'] = reason_text[:500]
# ProxMenux Host Backup: pull the extra fields that the runner
# packs into payload.fields so the host_backup_* templates can
# render backend, destination, sizes, etc. without falling back
# to '{field}' literals or empty placeholders.
if pve_type.startswith('proxmenux-host-backup-'):
backend = (fields.get('backend') or '').strip().lower()
backend_label_map = {
'pbs': 'Proxmox Backup Server',
'local': 'Local archive',
'borg': 'Borg repository',
}
data['backend'] = backend
data['backend_label'] = backend_label_map.get(backend, backend or 'unknown')
data['destination'] = fields.get('destination', '') or ''
data['profile_mode'] = fields.get('profile_mode', 'default') or 'default'
data['data_size'] = fields.get('data_size', '') or '-'
data['archive_size'] = fields.get('archive_size', '') or '-'
data['duration'] = fields.get('duration', '') or '-'
data['log_file'] = fields.get('log_file', '') or ''
# `reason` is what the body template substitutes for the
# failure case; fall back to the human message the runner
# sent so the operator never sees an empty Reason line.
data['reason'] = fields.get('reason', message) or message
# ── Extract clean reason for system-mail events ──
# smartd and other system mail contains verbose boilerplate.
@@ -3875,7 +4103,14 @@ class ProxmoxHookWatcher:
dur_m = re.search(r'Total running time:\s*(.+?)(?:\n|$)', message)
if dur_m:
data['duration'] = dur_m.group(1).strip()
# Extract storage / destination from the "starting new backup job" line
# so the notification (email HTML, Telegram title) can show which
# PBS / storage the backup landed on. Operators with multiple PBS
# targets need this to diagnose which destination failed.
storage_m = re.search(r'--storage\s+(\S+)', message)
if storage_m:
data['storage'] = storage_m.group(1).strip()
# Capture journal context for critical/warning events (helps AI provide better context)
if severity in ('CRITICAL', 'WARNING') and event_type not in ('backup_complete', 'update_available'):
# Build keywords from available data for journal search
@@ -3923,6 +4158,21 @@ class ProxmoxHookWatcher:
if severity in ('error', 'err'):
return 'backup_fail', 'vm', ''
return 'backup_complete', 'vm', ''
# ProxMenux Host Backups run as a standalone systemd timer +
# `run_scheduled_backup.sh` (or manually via `backup_host.sh`),
# NOT as PVE vzdump tasks. Routed to dedicated event types
# (`host_backup_*`) so the operator can toggle them separately
# from VM/CT backup events and the body can show host-backup
# specifics (backend, destination, data/archive size). Without
# this branch the Host Backup ran fine but never produced a
# notification — operator-reported.
if pve_type == 'proxmenux-host-backup-start':
return 'host_backup_start', 'node', ''
if pve_type == 'proxmenux-host-backup-complete':
return 'host_backup_complete', 'node', ''
if pve_type == 'proxmenux-host-backup-fail':
return 'host_backup_fail', 'node', ''
if pve_type == 'fencing':
return 'split_brain', 'node', ''
+103 -34
View File
@@ -69,10 +69,17 @@ SENSITIVE_KEYS = {
'ai_api_key_anthropic',
'ai_api_key_openai',
'ai_api_key_openrouter',
'telegram.token',
# `telegram.bot_token` — was previously listed as `telegram.token`,
# which never matched the real config_key (`bot_token` in
# CHANNEL_TYPES). The mismatch silently sent Telegram bot tokens
# in cleartext on every GET /api/notifications/settings since the
# masking layer was introduced. Fixed alongside the eye-reveal
# endpoint so the operator can still inspect the value on demand.
'telegram.bot_token',
'gotify.token',
'discord.webhook_url',
'email.password',
'apprise.url',
'webhook_secret',
}
@@ -932,23 +939,57 @@ class NotificationManager:
def start(self):
"""Start the notification service in server mode.
Launches watchers and dispatch loop as daemon threads.
Detection (the polling collector + watchers) ALWAYS runs because
the managed_installs registry NVIDIA, ProxMenux, Coral, OCI
updates drives the dashboard's "update available" UI even when
notifications are off. Caught on .89 in June 2026: the user had
disabled notifications back in May and the NVIDIA card was stuck
on a stale "v580.159.03 available" because the polling collector
was gated behind `self._enabled` here and never ran again.
Notification *delivery* (channel setup, cooldown reset, PVE
webhook, dispatch loop emitting events) stays conditional on
`self._enabled` the dispatch loop itself bails early when
disabled, so events from the watchers queue up briefly and get
dropped without ever being sent.
Called by flask_server.py on startup.
"""
if self._running:
return
self._load_config()
self._load_cooldowns_from_db()
if not self._enabled:
print("[NotificationManager] Service is disabled. Skipping start.")
return
self._running = True
self._stats['started_at'] = datetime.now().isoformat()
# ── Detection (always on) ────────────────────────────────
# Even when notifications are disabled, these watchers and the
# polling collector keep the managed_installs registry, the
# error history, and the task state up to date.
self._journal_watcher = JournalWatcher(self._event_queue)
self._task_watcher = TaskWatcher(self._event_queue)
self._polling_collector = PollingCollector(self._event_queue)
self._journal_watcher.start()
self._task_watcher.start()
self._polling_collector.start()
# Dispatch loop runs unconditionally too; its internal
# `if not self._enabled` guards drop events when disabled.
# Without it the event queue would grow forever.
self._dispatch_thread = threading.Thread(
target=self._dispatch_loop, daemon=True, name='notification-dispatch'
)
self._dispatch_thread.start()
if not self._enabled:
print("[NotificationManager] Notifications disabled — detection on, delivery off.")
return
# ── Delivery setup (only when enabled) ────────────────────
# Reset cooldowns for the curated event-type set so the user gets
# a fresh status report (update_summary, …) and a fresh security
# signal (auth_fail) after every Monitor deploy/restart. The 24h
@@ -971,22 +1012,7 @@ class NotificationManager:
pass # flask_notification_routes not loaded yet (early startup)
except Exception as e:
print(f"[NotificationManager] PVE webhook setup error: {e}")
# Start event watchers
self._journal_watcher = JournalWatcher(self._event_queue)
self._task_watcher = TaskWatcher(self._event_queue)
self._polling_collector = PollingCollector(self._event_queue)
self._journal_watcher.start()
self._task_watcher.start()
self._polling_collector.start()
# Start dispatch loop
self._dispatch_thread = threading.Thread(
target=self._dispatch_loop, daemon=True, name='notification-dispatch'
)
self._dispatch_thread.start()
print(f"[NotificationManager] Started with channels: {list(self._channels.keys())}")
def stop(self):
@@ -1514,7 +1540,22 @@ class NotificationManager:
def _flush_digest_for_channel(self, ch_name: str, channel: Any,
now: datetime) -> None:
"""Read pending rows for the channel, render a grouped summary,
send it, and delete the buffer entries on success."""
send it, and delete the buffer entries on success.
Every path through here records a `digest` entry in the
notification history (success or fail, empty buffer or not) and
bumps `_stats['total_sent']` / `total_errors` accordingly the
rest of the notification system does this in
`_dispatch_to_channels`, and skipping it here was the reason the
operator's `/api/notifications/history` and `total_sent` counter
showed zero digest entries even when the schedule was firing
(issue #233).
"""
host = _hostname(self._config)
summary_title = (
f"{host}: 24h summary ({now.strftime('%Y-%m-%d %H:%M')})"
)
try:
conn = sqlite3.connect(str(DB_PATH), timeout=10)
conn.execute('PRAGMA journal_mode=WAL')
@@ -1528,6 +1569,12 @@ class NotificationManager:
conn.close()
except Exception as e:
print(f"[NotificationManager] digest read failed for {ch_name}: {e}")
self._record_history(
'digest', ch_name, summary_title,
f'digest read failed: {e}', 'INFO',
False, str(e), 'digest_scheduler',
)
self._stats['total_errors'] += 1
return
# Mark `last_at` even if there's nothing to send — otherwise an
@@ -1535,24 +1582,46 @@ class NotificationManager:
self._save_setting(f'{ch_name}.digest_last_at', now.isoformat())
if not rows:
# Empty digest: don't ping the channel (no point in sending
# "nothing to report"), but DO log the run in history so the
# operator can see the schedule fired. Without this the digest
# looked dead silent when in fact it was working — there was
# just nothing INFO non-exempt to summarize.
self._record_history(
'digest', ch_name, summary_title,
'No INFO events buffered for this digest window.',
'INFO', True, '', 'digest_scheduler',
)
return
host = _hostname(self._config)
summary_title = (
f"{host}: 24h summary ({now.strftime('%Y-%m-%d %H:%M')})"
)
summary_body = self._compose_digest_body(rows)
result: dict = {'success': False, 'error': ''}
try:
channel.send(summary_title, summary_body, severity='INFO',
data={'_digest': True, '_count': len(rows)})
result = channel.send(summary_title, summary_body, severity='INFO',
data={'_digest': True, '_count': len(rows)}) or result
except Exception as e:
print(f"[NotificationManager] digest send failed for "
f"{ch_name}: {e}")
return
result = {'success': False, 'error': str(e)}
if result.get('success'):
self._stats['total_sent'] += 1
self._stats['last_sent_at'] = datetime.now().isoformat()
else:
self._stats['total_errors'] += 1
self._record_history(
'digest', ch_name, summary_title, summary_body, 'INFO',
result.get('success', False), result.get('error', '') or '',
'digest_scheduler',
)
# Delete only after a successful send so a transient failure
# doesn't lose the day's data.
# doesn't lose the day's data. The pre-fix version deleted
# unconditionally as long as `channel.send` didn't raise — a
# silent `{'success': False}` would still wipe the buffer.
if not result.get('success'):
return
try:
ids = [r[0] for r in rows]
conn = sqlite3.connect(str(DB_PATH), timeout=10)
+157 -25
View File
@@ -448,35 +448,44 @@ TEMPLATES = {
# status oscillation (OK->WARNING->OK) which creates noise.
# The health_persistent and new_error templates cover this better.
'state_change': {
'title': '{hostname}: {category} changed to {current}',
'title': '{hostname}: {category} changed to {current}{entity_suffix}',
'body': '{category} status changed from {previous} to {current}.\n{reason}',
'label': 'Health state changed',
'group': 'health',
'default_enabled': False,
},
'new_error': {
'title': '{hostname}: New {severity} - {category}',
'title': '{hostname}: New {severity} - {category}{entity_suffix}',
'body': '{reason}',
'label': 'New health issue',
'group': 'health',
'default_enabled': True,
},
'error_resolved': {
'title': '{hostname}: Resolved - {category}',
# `{entity}` is populated by health_persistence.resolve_error()
# (via _entity_from_details) and by PollingCollector's spread of
# the original details blob. When absent, _SafeDict elides the
# placeholder and the title collapses back to "Resolved - <cat>"
# without a trailing dash.
'title': '{hostname}: Resolved - {category}{entity_suffix}',
'body': 'The {category} issue has been resolved.\n{reason}\n\U0001F6A6 Previous severity: {original_severity}\n\u23F1\uFE0F Duration: {duration}',
'label': 'Recovery notification',
'group': 'health',
'default_enabled': True,
},
'error_escalated': {
'title': '{hostname}: Escalated to {severity} - {category}',
'title': '{hostname}: Escalated to {severity} - {category}{entity_suffix}',
'body': '{reason}',
'label': 'Health issue escalated',
'group': 'health',
'default_enabled': True,
},
'health_degraded': {
'title': '{hostname}: Health check degraded',
# flask_server._health_collector already builds the rich title
# (e.g. "prox: Storage CRITICAL — Tuxis (dir)") and passes it in
# data['title']. Fall back to the constant when absent so hand-
# rolled callers keep working.
'title': '{title_or_default}',
'body': '{reason}',
'label': 'Health check degraded',
'group': 'health',
@@ -637,26 +646,57 @@ TEMPLATES = {
'default_enabled': False,
},
'backup_complete': {
'title': '{hostname}: Backup complete — {vmname} ({vmid})',
'body': 'Backup of {vmname} (ID: {vmid}) completed successfully.\nSize: {size}',
'title': '{hostname}{storage}: Backup complete — {vmname} ({vmid})',
'body': 'Backup of {vmname} (ID: {vmid}) completed successfully on {storage}.\nSize: {size}',
'label': 'Backup complete',
'group': 'backup',
'default_enabled': True,
},
'backup_warning': {
'title': '{hostname}: Backup complete with warnings — {vmname} ({vmid})',
'body': 'Backup of {vmname} (ID: {vmid}) completed but encountered warnings.\nWarnings: {reason}',
'title': '{hostname}{storage}: Backup complete with warnings — {vmname} ({vmid})',
'body': 'Backup of {vmname} (ID: {vmid}) on {storage} completed but encountered warnings.\nWarnings: {reason}',
'label': 'Backup (warnings)',
'group': 'backup',
'default_enabled': True,
},
'backup_fail': {
'title': '{hostname}: Backup FAILED — {vmname} ({vmid})',
'body': 'Backup of {vmname} (ID: {vmid}) failed.\nReason: {reason}',
'title': '{hostname}{storage}: Backup FAILED — {vmname} ({vmid})',
'body': 'Backup of {vmname} (ID: {vmid}) failed on {storage}.\nReason: {reason}',
'label': 'Backup FAILED',
'group': 'backup',
'default_enabled': True,
},
# ── ProxMenux Host Backup events ──
# Distinct event types from `backup_*` (which are for vzdump VM/CT
# backups). The runner — both `run_scheduled_backup.sh` (timer) and
# the manual flow from `backup_host.sh` — POSTs to the local webhook
# with these explicit pve_type values, so the operator can toggle
# Host Backup notifications separately from VM/CT backup ones.
# Backend label (PBS / local / Borg) is highlighted in the title so
# the operator can tell at a glance where the data went.
'host_backup_start': {
'title': '{hostname}: Host backup started → {backend_label}',
'body': 'Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nProfile: {profile_mode}',
'label': 'Host backup started',
'group': 'backup',
'default_enabled': False,
},
'host_backup_complete': {
'title': '{hostname}: Host backup complete → {backend_label}',
'body': 'Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nData size: {data_size}\nArchive size: {archive_size}\nDuration: {duration}',
'label': 'Host backup complete',
'group': 'backup',
'default_enabled': True,
},
'host_backup_fail': {
'title': '{hostname}: Host backup FAILED → {backend_label}',
'body': 'Job: {job_id}\nBackend: {backend_label}\nDestination: {destination}\nDuration before failure: {duration}\nReason: {reason}\nLog: {log_file}',
'label': 'Host backup FAILED',
'group': 'backup',
'default_enabled': True,
},
'snapshot_complete': {
'title': '{hostname}: Snapshot created — {vmname} ({vmid})',
'body': 'Snapshot "{snapshot_name}" created for {vmname} (ID: {vmid}).',
@@ -756,7 +796,7 @@ TEMPLATES = {
'default_enabled': True,
},
'pci_passthrough_conflict': {
'title': '{hostname}: PCIe device conflict detected',
'title': '{hostname}: PCIe device conflict detected{device_pci}',
'body': (
'A PCIe device is assigned to multiple guests.\n'
'Device: {device_pci}\n'
@@ -778,7 +818,7 @@ TEMPLATES = {
# ── Network events ──
'network_down': {
'title': '{hostname}: Network connectivity lost',
'title': '{hostname}: Network connectivity lost{entity_suffix}',
'body': 'The node has lost network connectivity.\nReason: {reason}',
'label': 'Network connectivity lost',
'group': 'network',
@@ -808,7 +848,7 @@ TEMPLATES = {
'default_enabled': True,
},
'firewall_issue': {
'title': '{hostname}: Firewall issue detected',
'title': '{hostname}: Firewall issue detected{entity_suffix}',
'body': 'A firewall configuration issue has been detected.\nReason: {reason}',
'label': 'Firewall issue detected',
'group': 'security',
@@ -868,8 +908,24 @@ TEMPLATES = {
'group': 'services',
'default_enabled': True,
},
'system_restore_completed': {
'title': '{hostname}: Host restore finished',
'body': (
'Post-restore tasks completed in background.\n\n'
'Guests applied: {guests}\n'
'Bind-mount stubs: {stubs}\n'
'Stale node dirs removed: {stale_nodes}\n'
'Components reinstalled: {components}\n'
'Duration: {duration}\n'
'{warnings_block}\n'
'The node is now fully ready to use.'
),
'label': 'Host restore completed',
'group': 'services',
'default_enabled': True,
},
'system_problem': {
'title': '{hostname}: System problem detected',
'title': '{hostname}: System problem detected{entity_suffix}',
'body': 'A system-level problem has been detected.\nReason: {reason}',
'label': 'System problem detected',
'group': 'services',
@@ -892,7 +948,7 @@ TEMPLATES = {
# ── Hidden internal templates (not shown in UI) ──
'service_fail_batch': {
'title': '{hostname}: {service_count} services failed',
'title': '{hostname}: {service_count} services failed{entity_suffix}',
'body': '{reason}',
'label': 'Service fail batch',
'group': 'services',
@@ -957,31 +1013,31 @@ TEMPLATES = {
'hidden': True,
},
'unknown_persistent': {
'title': '{hostname}: Check unavailable - {category}',
'title': '{hostname}: Check unavailable - {category}{entity_suffix}',
'body': 'Health check for {category} has been unavailable for 3+ cycles.\n{reason}',
'label': 'Check unavailable',
'group': 'health',
'default_enabled': False,
'hidden': True,
},
# ── Health Monitor events ──
'health_persistent': {
'title': '{hostname}: {count} active health issue(s)',
'title': '{hostname}: {count} active health issue(s){entity_suffix}',
'body': 'The following health issues remain unresolved:\n{issue_list}\n\nThis digest is sent once every 24 hours while issues persist.',
'label': 'Active health issues (daily)',
'group': 'health',
'default_enabled': True,
},
'health_issue_new': {
'title': '{hostname}: New health issue — {category}',
'title': '{hostname}: New health issue — {category}{entity_suffix}',
'body': 'New {severity} issue detected in: {category}\nDetails: {reason}',
'label': 'New health issue',
'group': 'health',
'default_enabled': True,
},
'health_issue_resolved': {
'title': '{hostname}: Resolved - {category}',
'title': '{hostname}: Resolved - {category}{entity_suffix}',
'body': '{category} issue has been resolved.\n{reason}\nDuration: {duration}',
'label': 'Health issue resolved',
'group': 'health',
@@ -1091,7 +1147,7 @@ TEMPLATES = {
'title': '{hostname}: CT {vmid} rootfs at {usage_percent}%',
'body': (
'CT {vmid} ({name}) rootfs is at {usage_percent}% '
'({disk_bytes} / {maxdisk_bytes}).\n\n'
'({disk_bytes_human} / {maxdisk_bytes_human}).\n\n'
'A full LXC rootfs prevents the container from booting cleanly. '
'Either expand the rootfs (pct resize {vmid} rootfs +1G) or free '
'space inside the container.'
@@ -1177,11 +1233,19 @@ TEMPLATES = {
# bullet list make sure the operator sees exactly what's moving
# without opening the dashboard first.
'secure_gateway_update_available': {
'title': '{hostname}: {app_name} update available — v{latest_version}',
# `{update_title_suffix}` and `{version_line}` are computed in
# `_build_managed_install_event` so the same template can render
# both scenarios cleanly:
# - Tailscale itself moved: "— v<new>" + "current → latest" line
# - Only sidecar packages moved: "— v<current> (packages only)" +
# single-line "Tailscale unchanged" body
# Prevents the confusing "v1.90.9-r6 → v1.90.9-r6" render when
# Tailscale is stable and only alpine libs are updating.
'title': '{hostname}: {app_name} update available{update_title_suffix}',
'body': (
'{app_name} (managed by ProxMenux) has 📦 {package_count} package update(s) '
'pending in its container.\n'
'🔹 Current Tailscale: v{current_version} → 🟢 Latest: v{latest_version}\n\n'
'{version_line}\n\n'
'💡 Open ProxMenux Monitor > Settings > Secure Gateway and click '
'"Update" to apply.\n\n'
'🗂️ Packages:\n{package_list}'
@@ -1338,6 +1402,28 @@ def _get_hostname() -> str:
return 'proxmox'
def _format_bytes_human(n: Any) -> str:
"""Compact human-readable size (base 1024): '35.3 GiB', '4.8 TiB'.
Matches what `df -h` and the Proxmox UI show, so operators reading a
notification aren't asked to translate a 12-digit byte count.
"""
try:
size = float(n)
except (TypeError, ValueError):
return ''
if size <= 0:
return '0 B'
units = ('B', 'KiB', 'MiB', 'GiB', 'TiB', 'PiB')
i = 0
while size >= 1024.0 and i < len(units) - 1:
size /= 1024.0
i += 1
if i == 0:
return f'{int(size)} B'
return f'{size:.1f} {units[i]}'
def render_template(event_type: str, data: Dict[str, Any]) -> Dict[str, Any]:
"""Render a template into a structured notification object.
@@ -1381,12 +1467,54 @@ def render_template(event_type: str, data: Dict[str, Any]) -> Dict[str, Any]:
'issue_list': '', 'error_key': '',
'storage_name': '', 'storage_type': '',
'important_list': 'none',
# Host Backup specifics (run_scheduled_backup.sh + backup_host.sh).
'job_id': '', 'backend': '', 'backend_label': '',
'destination': '', 'profile_mode': '',
'data_size': '', 'archive_size': '',
'log_file': '',
}
variables.update(data)
# Humanise raw-byte fields so templates can render '35.3 GiB' instead
# of '37952020480'. Producers keep emitting raw ints (needed by APIs
# and the dashboard); the humanised twin is derived on the fly here.
for _byte_key in ('disk_bytes', 'maxdisk_bytes'):
if _byte_key in data:
variables[f'{_byte_key}_human'] = _format_bytes_human(data[_byte_key])
# Ensure important_list is never blank (fallback to 'none')
if not variables.get('important_list', '').strip():
variables['important_list'] = 'none'
# Derive the affected object's display name for titles that use it.
# Priority: caller-supplied `entity` (health_monitor.emit_event) →
# per-detail fields spread in from `record_error`'s details blob.
# `entity_suffix` gives templates a way to render "…- Storage 'Tuxis'"
# without the trailing " - " when the field is empty.
_ent = str(variables.get('entity', '')).strip()
if not _ent:
_ent = (
variables.get('storage_name')
or variables.get('mount_point')
or variables.get('device')
or variables.get('interface')
or variables.get('vm_name')
or ''
)
_ent = str(_ent).strip()
variables['entity'] = _ent
variables['entity_suffix'] = f'{_ent}' if _ent else ''
# `title_or_default` lets a template surface a caller-computed title
# when the caller already knows the entity context (e.g.
# flask_server's health collector), and fall back to a stock string
# otherwise. Prevents empty subject lines when the caller forgot to
# populate `title`.
_caller_title = str(variables.get('title', '')).strip()
if not _caller_title:
hn = variables.get('hostname', '')
_caller_title = f'{hn}: Health check degraded' if hn else 'Health check degraded'
variables['title_or_default'] = _caller_title
# `format_map` with a SafeDict avoids the KeyError → "show raw template
# with `{placeholder}` literal" failure mode. If a template gets a new
@@ -1567,6 +1695,9 @@ EVENT_EMOJI = {
'replication_complete': '\u2705',
# Backups
'backup_start': '\U0001F4BE\U0001F680', # 💾🚀 floppy + rocket
'host_backup_start': '\U0001F5C4\U0001F680', # 🗄️🚀 cabinet + rocket
'host_backup_complete': '\U0001F5C4️✅', # 🗄️✅ cabinet + check
'host_backup_fail': '\U0001F5C4️❌', # 🗄️❌ cabinet + cross
'backup_complete': '\U0001F4BE\u2705', # 💾✅ floppy + check
'backup_warning': '\U0001F4BE\u26A0\uFE0F', # 💾⚠️ floppy + warning
'backup_fail': '\U0001F4BE\u274C', # 💾❌ floppy + cross
@@ -1604,6 +1735,7 @@ EVENT_EMOJI = {
'system_startup': '\U0001F680', # rocket (startup)
'system_shutdown': '\u23FB\uFE0F', # power symbol (Unicode)
'system_reboot': '\U0001F504',
'system_restore_completed': '', # check mark
'system_problem': '\u26A0\uFE0F',
'service_fail': '\u274C',
'oom_kill': '\U0001F4A3', # bomb
+9
View File
@@ -1508,6 +1508,15 @@ def check_app_update_available(app_id: str, force: bool = False) -> Dict[str, An
except Exception:
pass
# When Tailscale itself isn't in the update list but other packages
# inside the container are, `latest_version` stays None and downstream
# notification renders "— v" / "🟢 Latest: v" with a dangling prefix.
# Anchor it to the installed version so the notification reads
# cleanly ("current == latest" tells the operator Tailscale itself
# is unchanged, only sidecar packages are updating).
if result["current_version"] and not result["latest_version"]:
result["latest_version"] = result["current_version"]
_app_update_cache[app_id] = result
return result
+71 -5
View File
@@ -27,7 +27,8 @@ import re
import threading
import time
from pathlib import Path
from typing import Any
import os
from typing import Any, Callable
_BASE = Path("/usr/local/share/proxmenux")
_POST_INSTALL_DIR = _BASE / "scripts" / "post_install"
@@ -47,6 +48,9 @@ _FN_DEF_RE = re.compile(r"^(?P<name>[a-zA-Z_][a-zA-Z0-9_]*)\s*\(\)\s*\{\s*$")
_VERSION_RE = re.compile(r'local\s+FUNC_VERSION\s*=\s*"([0-9]+(?:\.[0-9]+)+)"')
_DESC_RE = re.compile(r"#\s*description\s*:\s*([^\n]+)")
_REGISTER_RE = re.compile(r'\bregister_tool\s+"([^"]+)"\s+true\b')
# Matches a heredoc opener and captures its terminator word. Handles
# `<<EOF`, `<<-EOF`, `<< EOF`, `<<'EOF'` and `<<"EOF"`.
_HEREDOC_RE = re.compile(r'<<[-~]?\s*["\']?([A-Za-z_][A-Za-z0-9_]*)["\']?')
# In-memory cache of the last scan. Sprint 12A uses a single startup scan
# plus on-demand re-scan via the API; no automatic refresh.
@@ -121,12 +125,31 @@ def parse_post_install_script(path: Path) -> dict[str, dict[str, str]]:
continue
func_name = match.group("name")
# Find the matching closing brace at column 0. Bash post-install
# scripts use the convention `}` on its own line at the start of
# the line to close top-level functions, so we scan until that.
# Find the matching closing brace at column 0. Top-level post-install
# functions close with `}` alone on a line. A bare scan for that line
# breaks when the body embeds a heredoc whose content also has `}` at
# column 0 — e.g. a logrotate stanza written via `cat >... <<EOF`.
# That truncated the body before its `register_tool` call, so tools
# like log2ram / network_optimization were never registered and never
# flagged as updatable. Track heredoc state and ignore `}` inside one.
body_start = i + 1
body_end = body_start
while body_end < len(lines) and not lines[body_end].rstrip() == "}":
heredoc_term = None
while body_end < len(lines):
current = lines[body_end]
if heredoc_term is not None:
# Inside a heredoc — only its terminator line ends it.
if current.strip() == heredoc_term:
heredoc_term = None
body_end += 1
continue
if current.rstrip() == "}":
break
# A line may open a heredoc (use the last opener on the line, so
# `cmd <<A | cmd <<B` picks B, matching shell behaviour).
openers = _HEREDOC_RE.findall(current)
if openers:
heredoc_term = openers[-1]
body_end += 1
body = "\n".join(lines[body_start:body_end])
@@ -191,9 +214,52 @@ def load_installed_tools(path: Path = _INSTALLED_JSON) -> dict[str, dict[str, An
else:
# Unknown shape — treat as not installed rather than crash.
normalized[key] = {"installed": False, "version": "", "source": ""}
_apply_legacy_detectors(normalized)
return normalized
# Tool keys that were added later than 1.2.2 and therefore are missing
# from installed_tools.json on hosts that already had the underlying
# config in place. For each, declare how to recognise "this host has
# the thing configured already" so the registry surfaces it as a v1.0
# legacy install — the per-tool wrapper then bumps it to the current
# FUNC_VERSION on first apply, and from then on the regular update
# detector takes over without any special-casing.
_LEGACY_DETECTORS: dict[str, Callable[[], bool]] = {
"proxmox_repos": lambda: any(
os.path.exists(p) for p in (
"/etc/apt/sources.list.d/proxmox.sources",
"/etc/apt/sources.list.d/pve-no-subscription.list",
"/etc/apt/sources.list.d/pve-enterprise.list",
)
),
}
def _apply_legacy_detectors(normalized: dict[str, dict[str, Any]]) -> None:
"""Inject synthetic v1.0 entries for tools that the host already
has configured but were never recorded in installed_tools.json.
Skips any tool that already has an entry once the user applies
the update, ``register_tool`` writes the real entry and the
detector path becomes a no-op for that key.
"""
for key, probe in _LEGACY_DETECTORS.items():
if key in normalized:
continue
try:
if probe():
normalized[key] = {
"installed": True,
"version": "1.0",
"source": "detected",
}
except Exception:
# Detectors must never break the registry load.
continue
# ---------------------------------------------------------------------------
# Detection logic
# ---------------------------------------------------------------------------
+19 -5
View File
@@ -1,6 +1,10 @@
{
"compilerOptions": {
"lib": ["dom", "dom.iterable", "es6"],
"lib": [
"dom",
"dom.iterable",
"es6"
],
"allowJs": true,
"skipLibCheck": true,
"strict": true,
@@ -19,9 +23,19 @@
],
"baseUrl": ".",
"paths": {
"@/*": ["./*"]
}
"@/*": [
"./*"
]
},
"target": "ES2017"
},
"include": ["next-env.d.ts", "**/*.ts", "**/*.tsx", ".next/types/**/*.ts"],
"exclude": ["node_modules"]
"include": [
"next-env.d.ts",
"**/*.ts",
"**/*.tsx",
".next/types/**/*.ts"
],
"exclude": [
"node_modules"
]
}
+143 -12
View File
@@ -1,4 +1,135 @@
## 2026-07-22
### New version ProxMenux v1.2.4
This release adds two in-dashboard improvements — a one-click Proxmox update trigger from the Health Monitor and a mobile PWA install prompt — extends the Backups restore flow with atomic pmxcfs (`config.db`) snapshots and automatic ZFS data-pool import, sharpens Log2RAM behaviour on hosts running Proxmox Backup Server as a service, hardens firewall bridge sysctl tuning across VM lifecycle events, narrows the ZFS ARC optimization to its own scope, makes persistent NIC naming idempotent across reruns, rebuilds DKMS drivers automatically when a new kernel is staged, keeps the Monitor terminal session intact when a ProxMenux update is available, and reinforces five notification templates plus three Health panel checks.
---
## 🩺 Update Now button in Health Monitor
- New **Update Now** button inside the Health Monitor modal, under the **System Updates** section.
- Runs the standard Proxmox update flow (`apt update` + `dist-upgrade` + post-update cleanup) in an in-dashboard terminal — no need to open a shell.
- Only appears when updates are pending; when the system is up to date the button stays hidden.
- On close, the Health Monitor forces a cache-busting refresh (`/api/health/full?refresh=1`) so the pending-update count and kernel row reflect the post-update state right away, instead of the pre-update value that the background cache had stored moments before the run.
- The underlying script is context-aware: on an already-configured production host it respects the user's custom repositories (never disables enterprise/ceph, never deletes legacy sources, never purges alternate NTP, never force-installs zfsutils/chrony); on a bare host it lays down only the missing base repos. It also detects a newly installed kernel that isn't the running one and prompts for reboot at the end.
- During the upgrade, `service_fail` notifications for PVE services (pve-cluster, pveproxy, corosync…) are suppressed — their restart is a normal part of the upgrade cycle. Suppression extends 60 s past apt exit so the trailing restart events don't leak through.
---
## 📱 In-app Install prompt for mobile
- First-time visitors on **Android** (Chrome / Brave) and **iOS Safari** now see a bottom-sheet with clear instructions for adding the Monitor as a PWA to their home screen.
- Installation goes through the browser's own menu entry ("Add to Home Screen"), which produces a real installed PWA that launches in standalone mode. The sheet doesn't intercept the browser's `beforeinstallprompt` event — intercepting it and not calling `prompt()` degrades the manual menu path to a plain shortcut, which is what showed up in field testing.
- Two dismissal levels: **Not now** (temporary, reappears in 30 days) and **Don't show again** (permanent, stored in `localStorage`).
- Never shown on desktop, or once the Monitor is already running standalone.
---
## 🔔 Notification content — five rendering refinements
- **Backup destination in title and body** — VM/CT backup emails and Telegram messages carry the storage / PBS target, so users with several backup destinations can tell at a glance which one produced the event.
- **Migration bodies carry the real target node** — pulled from the PVE task log for `qmigrate` / `vzmigrate` events.
- **Snapshot bodies carry the real snapshot name** — pulled from the PVE task log for `qmsnapshot` / `vzsnapshot` events.
- **Generic `system_problem` notifications include the real reason** — PVE payload messages are surfaced as the notification body.
- **NVIDIA / Coral driver update emails render the *New Version* row correctly** — the template placeholder is now aligned with the field the renderer reads.
---
## 🩹 Health panel — three checks reinforced
- **Dismiss now silences storage alerts.** The acknowledge flow includes `storage_unavailable`, `mount_stale`, `mount_readonly`, `lxc_disk_low`, `lxc_mount_low`, `pve_storage_full` and `zfs_pool_full` under the `storage` category, and the storage cache is invalidated on dismiss so the panel refreshes immediately.
- **VMs & Containers check tolerates persisted errors with a NULL `details` column** (#255). `_check_vms_cts_with_persistence` coalesces missing `details` to an empty dict before reading nested keys, so a single sparse row no longer takes the whole VM/CT check offline.
- **`system_startup` notification fires once per boot.** `_check_startup_aggregation` marks aggregation as done right after queuing the event, so the boot summary lands one time regardless of how many polling ticks fit inside the session.
---
## 🛠 Mobile & webhook
- **Mobile dashboard polling stays live on HTTPS + reverse proxy setups.** `pwa-register.tsx` auto-unregisters any Service Worker on load so mobile-browser background throttling stops interfering with the polling fetches, and PWA installability is now driven by the new in-app install prompt above.
- **Webhook auth trusts every host-local IP.** The internal webhook (`/api/notifications/webhook`) accepts requests from any interface IP the host owns (Tailscale, WireGuard, LAN, IPv6, plus IPv4-mapped-in-IPv6 form `::ffff:x.x.x.x` that Flask emits on dual-stack binds), so PVE Test buttons work through any of them.
---
## 🛡 Update flow — Monitor-terminal-aware for update and channel switch
- **The Monitor's WebSocket terminal now exposes `PROXMENUX_TERMINAL=monitor`** in the environment of every shell it opens, and every child inherits it. This gives `menu` (and any other flow that cares) a reliable, deterministic way to tell that the current session lives inside the Monitor process — a session that would be cut mid-install if the Monitor service was restarted.
- **`menu` update prompt** — when a new ProxMenux version is available and the session is running inside the Monitor terminal, the classic yes/no update prompt is replaced by an informational msgbox. The msgbox names the new version and shows the canonical one-liner (`bash -c "$(wget -qLO - …)"`) to run the update from SSH or the Proxmox host console. Because the flow has already decided the in-terminal update path is unsafe (see the [msgbox-ack rule](memory/feedback_whiptail_msgbox_ack.md)), there's a single OK button — no yes/no that could trigger the destructive update by accident.
- **Settings → Release Channel** — the same guard is applied in `config_menu.sh`'s `apply_release_channel()`. Selecting Stable ↔ Beta from the Monitor terminal shows an informational msgbox with the exact `wget` one-liner for the target channel (using the same URL the flow would have downloaded itself) and returns to the menu instead of running the installer in place.
- **After OK, both flows continue normally**. The user keeps using ProxMenux from the same terminal without restrictions; only the destructive step is routed elsewhere. There is no lockdown and no forced action.
- **SSH sessions, the Proxmox host console, and any environment where `PROXMENUX_TERMINAL` isn't `monitor` keep the previous behaviour** and can update or switch channels as always. The change only affects the case where doing it in place would break the running session.
- **Bootstrap note**: because `PROXMENUX_TERMINAL=monitor` is added by the AppImage this release ships, the guard only starts protecting sessions once the host is on 1.2.4 or newer. The very first update to 1.2.4, if triggered from the Monitor terminal, can still hit the old behaviour — from 1.2.4 forward the guard is in place.
---
## 🔧 Update flow — DKMS drivers rebuilt when a new kernel lands
- **After `apt full-upgrade` stages a kernel newer than the one currently running, `update-pve-safe.sh` now rebuilds ProxMenux-installed DKMS drivers against the new kernel.** The Update Now button in the Health Monitor and the `utilities/proxmox_update.sh` CLI both delegate to `update-pve-safe.sh`, so both routes gain the behaviour. The step reads `components_status.json`, cross-references the DKMS-managed components ProxMenux tracks (`nvidia_driver`, `coral_driver`), installs the matching kernel headers (`proxmox-headers-<newkver>` or `pve-headers-<newkver>`) if they aren't already present, and runs `dkms autoinstall -k <newkver>`. Then it verifies via `dkms status` that each expected module (`gasket` for Coral, `nvidia` for the NVIDIA driver) actually reached `installed` state for the new kernel — if any module didn't, it falls back to each installer's `--auto-reinstall` path.
- **A whiptail msgbox announces the rebuild before it runs.** Single OK button — no yes/no. Names the incoming kernel version and lists the DKMS components that are going to be rebuilt, so the user sees exactly what's about to happen. Because leaving DKMS drivers unbuilt would leave the system with a working kernel but non-functional TPU / GPU at boot, this is transparency, not a decision — pressing OK acknowledges the follow-up work and the flow proceeds. Non-interactive invocations (cron, headless batch, missing whiptail) skip the msgbox and log the same information.
- **Only components already registered as `installed` in `components_status.json` are considered.** A host with no ProxMenux-managed DKMS drivers sees no msgbox and no rebuild step. Hosts that never ran the Coral or NVIDIA installer are unaffected.
- **Failure to rebuild does not abort the update.** If a DKMS module can't be rebuilt against the new kernel (upstream API break, missing dependency), the update flow completes normally, the specific components that failed are named in the summary, and the user can re-run their installer manually after reboot. The step is best-effort by design — a kernel/driver mismatch is an upstream problem, not something the update flow should fail on.
- Shared helper `pmx_rebuild_dkms_after_kernel` lives in `scripts/global/utils-install-functions.sh`, so future updaters or CLI utilities can pick it up with a one-line call.
---
## 🔌 Post-install — Persistent NIC naming becomes idempotent
- **ProxMenux-owned `.link` files now carry a distinctive filename prefix and internal marker.** Files are written as `10-proxmenux-<iface>.link` and the first line of every file is `# Managed by ProxMenux — do not edit`. Both are checked by the reconciliation and uninstall paths before touching a file, so anything the user wrote by hand or that came from another package is safe.
- **Reruns of `setup_persistent_network` reconcile ProxMenux entries.** Every invocation walks the existing `10-proxmenux-*.link` files, extracts the `MACAddress=` value, compares it against the MACs currently present under `/sys/class/net/`, and removes only the ProxMenux-owned entries whose MAC is no longer there. Hardware replacements, NIC swaps and hardware migrations stop leaving orphan mappings behind on every rerun.
- **Legacy 1.0-format files (`10-<iface>.link` written by the previous revision) are migrated on the first run of the new function.** If the file matches the exact template the 1.0 code used to write (two sections, `MACAddress=` + `Name=`, nothing else), it's removed and replaced with the new `10-proxmenux-<iface>.link` in one step. Any file that doesn't match the template exactly is left alone.
- **The uninstall path (`uninstall_persistent_network`) now only removes files that carry both the `10-proxmenux-` filename prefix and the marker on the first line.** The previous `rm -f /etc/systemd/network/*.link` blanket sweep is gone — user-authored `.link` files stay in place regardless of their filename.
- **Single shared implementation.** The three duplicated `setup_persistent_network` bodies (`auto_post_install.sh`, `customizable_post_install.sh`, `network_menu.sh`) plus the uninstall path now all delegate to `pmx_setup_persistent_network` / `pmx_uninstall_persistent_network` in `scripts/global/utils-install-functions.sh`. Future fixes can't miss a copy.
- `FUNC_VERSION` bumped 1.0 → 1.1 on all three call sites so the ProxMenux update detector re-runs the function on hosts that already had the 1.0 build. That first re-run performs the legacy migration + reconciliation in one shot.
---
## 🧮 Post-install — ZFS ARC optimization narrowed to its scope
- **`optimize_zfs_arc` now sets only `zfs_arc_max`.** The function writes a single line to `/etc/modprobe.d/99-zfsarc.conf`: `options zfs zfs_arc_max=<cap>`. `zfs_arc_min` stays at the OpenZFS default (auto-calculated as the larger of 32 MiB and ~1/32 of RAM), and L2ARC (`l2arc_noprefetch`, `l2arc_write_max`) and TXG (`zfs_txg_timeout`) tunables — which are outside the scope of an ARC optimization — are left at their OpenZFS defaults unless the user configures them elsewhere.
- **The initramfs is now regenerated after writing the config.** On ZFS-on-root systems the ZFS module loads from the initramfs before the running system reads `/etc/modprobe.d/`, so a plain reboot wasn't enough for the new cap to take effect. `update-initramfs -u -k all` runs right after the file is written, plus `proxmox-boot-tool refresh` on systemd-boot hosts, so the value is picked up at the next boot instead of being shadowed by the initramfs's stale copy.
- **The function is guarded on the presence of a live ZFS pool** (`zpool list` check) so it becomes a no-op on hosts that don't use ZFS.
- **Cap values use clean binary sizes**: 512 MiB up to 16 GB RAM, 1 GiB up to 32 GB, RAM/8 above that — with a floor of 512 MiB so a bad memory reading never leaves an unusably small ARC.
- `FUNC_VERSION` bumped 1.0 → 1.1 so the ProxMenux update detector re-runs the function on hosts that already had the 1.0 build. Because the write is a full rewrite of `99-zfsarc.conf`, running the updated function once replaces the whole file cleanly. The uninstall path now also runs `update-initramfs` + `proxmox-boot-tool refresh` after restoring or removing the config, so the revert propagates to the initramfs the same way.
---
## 🔥 Post-install — Firewall bridge sysctl tuning hardened
- **The `rp_filter=0` and `log_martians=0` tuning for `fwbr*`, `fwln*`, `fwpr*` and `tap*` interfaces now also applies to interfaces Proxmox spins up when a VM starts, stops, reboots or migrates.** A new `/etc/udev/rules.d/99-proxmenux-fwbr-tune.rules` fires a helper on every `net`/`add` event matching those prefixes, so each fresh interface picks up the correct value immediately — no reboot and no rerun of the post-install needed.
- **The tuning logic is now in a standalone helper** at `/usr/local/sbin/proxmenux-fwbr-tune`, shared by the initial sweep (`proxmenux-fwbr-tune.service`, oneshot) and by the udev rule. An explicit invocation at install time ensures the current session sees the change without waiting for the next VM cycle.
- **The customizable post-install flow (`customizable_post_install.sh`) now installs the same helper + oneshot service + udev rule + initial sweep as the automatic flow**, so both variants leave the system in the same end state.
- Both `apply_network_optimizations` functions bumped `FUNC_VERSION` 1.0 → 1.1, so the ProxMenux update detector re-runs the function on hosts that already had the 1.0 build. The uninstall path (`uninstall_network_optimization`) is extended to remove the new helper and udev rule, and reload udev.
---
## 🧰 Post-install — Log2RAM + PBS
- **PBS API log rotation applied automatically when `proxmox-backup-server` runs as a service on the host.** Both Log2RAM installers (`install_log2ram_auto` and the customizable `configure_log2ram`) detect PBS via `dpkg-query` and drop `/etc/logrotate.d/proxmox-backup-api` with a 20MB × 3 rotation rule plus `/etc/cron.hourly/proxmox-backup-logrotate`. On a PVE host that also runs PBS as a service, `pvestatd`'s local-datastore poll writes to `/var/log/proxmox-backup/api/access.log` and `auth.log` every few seconds — the upstream PBS package ships no logrotate rule for those files, and this rule keeps them bounded so a tmpfs-backed `/var/log` stays comfortably under budget. No-op on hosts without PBS as a service.
- **Upstream `log2ram` script patched to `rsync -aXv --no-acls` right after `install.sh`.** Both installers rewrite the call in place with a `sed` guarded by `grep -q` (backup at `.proxmenux.bak`, no-op if a future upstream release already dropped `-A`). Extended attributes (`-X`) are preserved. Result: `log2ram write` finishes cleanly on `/var/log.hdd` filesystems that don't accept POSIX ACLs (ZFS with `acltype=off`, ext4 mounted without the `acl` option) — no more `set_acl: Operation not supported` / exit 23 messages.
- **Emergency block of `log2ram-check.sh` rotates PBS logs before truncating.** When `/var/log` crosses the 92% threshold, the auto-sync script now runs `logrotate -f /etc/logrotate.d/proxmox-backup-api` (only if the rule file exists) *before* truncating `pveproxy/access.log`, `pveproxy/error.log` and `pveam.log`. Recent PBS access/auth history is preserved in the rotated `.gz` files instead of being lost. Both `install_log2ram_auto` and `configure_log2ram` bumped `FUNC_VERSION` 1.2 → 1.3, and the embedded `log2ram-check.sh` header comment bumped v1.2 → v1.3.
## 🗄 Backup restore — pmxcfs snapshot + ZFS data pools
- **`/var/lib/pve-cluster/config.db` is now captured with `sqlite3 .backup`.** pmxcfs (`/etc/pve`) is served by `pve-cluster` from that SQLite store, so a plain rsync of the raw file with the service running can catch it mid-WAL checkpoint and land in the archive as an inconsistent copy. `hb_prepare_staging` now runs `sqlite3 /var/lib/pve-cluster/config.db ".backup '$staging/…/config.db'"` before the general rsync — the canonical way (documented by Proxmox) to snapshot the store consistently while `pve-cluster` keeps serving traffic, with zero downtime for the cluster. The general rsync of `/var/lib/pve-cluster` now excludes `config.db`, `config.db-wal` and `config.db-shm` so nothing overwrites the atomic dump. Hosts without `sqlite3` fall back to a raw copy named `config.db.raw-fallback`, which the recovery helper promotes to `config.db` before starting `pve-cluster`. Metadata records which path was used via `pmxcfs_config_db=sqlite_backup|raw_fallback` in `metadata/run_info.env` for trace. The restore path continues to use the canonical `systemctl stop pve-cluster → cp → systemctl start pve-cluster` pattern (`apply_pending_restore.sh` and the standalone recovery helper written next to every extracted cluster dir), so the DB the user brings back is now guaranteed consistent instead of a raw file copy of state in flight.
- **Separate ZFS data pools listed in the backup are now imported automatically at restore time.** The new `_rs_import_data_pools` step runs after config apply, walks `storage_inventory.zfs_pools[]`, skips the root pool (already mounted by the system), and issues `zpool import <name>` for every non-root pool whose disks are all present on this host. When ZFS rejects the import as *foreign* — the typical case after a fresh install regrabs the pool label with a new `hostid` — the step retries with `-f` and reports the pool as forced so the user has trace. Pools missing any disk are skipped with a clear warning rather than imported degraded. Together this closes the common case where `zfs-import-scan.service` failed at boot after a fresh install and left the separate data pool unavailable until `zpool import -f` was run manually.
- **The auto-import result persists to the post-restore progress card.** The step writes a `data_pools_import` section into `/var/lib/proxmenux/restore-state.json` (the same JSON the Backups-tab card polls) and a raw log at `/var/log/proxmenux/restore-datapools-<timestamp>.log`. The Backups tab card renders a dedicated block inside Details with five color-coded rows (Imported / Forced / Skipped partial / Skipped missing / Failed) so the summary stays consultable after the restore terminal is closed, and the entry is preserved in the run's history for later review.
- **ZFS pools created with `by-partuuid` or raw `/dev/sdX` are recognised by the disk-presence check.** The auto-import step and `validate_storage.sh` treat `devices_by_id` entries that start with `/` as absolute paths and only prepend `/dev/disk/by-id/` to bare basenames, so pools built against partition UUIDs or a raw block device are detected as present when their disks are on the host.
- **`/etc/systemd/network` added to the default backup paths.** That directory holds systemd `.link` files that pin NIC names to their MAC across kernel updates and reinstalls — `setup_persistent_network` in the post_install writes them for every physical interface, and users can drop their own to rename a NIC to something meaningful. Preserving them across a fresh-install restore keeps the source host's NIC naming policy intact on the target, so `/etc/network/interfaces` entries that reference custom NIC names continue to resolve after the restore.
---
## 🙏 Acknowledgments
- **@pepenai** — mobile dashboard on HTTPS + reverse proxy.
- **Pepo** — webhook auth from a Tailscale FQDN.
- **@ash34** (#255) — VM/CT check with a NULL `details` row.
- **@f3rs3n** (#256, #257, #258) — firewall bridge sysctl tuning, ZFS ARC optimization scope, and persistent NIC naming reconciliation.
- **Juan C.** — ZFS data pool auto-import after a fresh install.
- **David Barbero (@sikete)** — DKMS driver rebuild on kernel upgrade.
## 2026-07-14
### New version ProxMenux v1.2.3
@@ -17,7 +148,7 @@ Stable consolidation of the **v1.2.2.x beta cycle** (v1.2.2.1 → v1.2.2.2 → v
- **PBS encryption with recovery blob**: encrypted backups store a passphrase-wrapped copy of the keyfile as a `-keyrecovery` group next to each backup, so a new Proxmox install can always get the key back with the user's passphrase.
- **Direction-aware restore**: reapplies IOMMU / VFIO / GRUB tunings on cross-kernel jumps, protects critical packages from cascade-remove, auto-remaps NICs after a motherboard swap.
- **Live post-restore progress card**: after the reboot, the Backups tab shows a real-time card with step-by-step milestones, per-component status (NVIDIA, Intel GPU tools, Coral, AMD tools) and a log tail with an Issues-only filter. Past restores are archived and browsable.
- **PBS keyfile management inline in the Monitor**: each PBS destination row exposes Download / Upload / Delete for the keyfile plus a Yes/No + passphrase + contextual Apply toggle for the escrow. When the installed keyfile does not match the backup's manifest, View contents / Download / Restore now show a structured amber panel with the required fingerprint so the operator knows which keyfile to import.
- **PBS keyfile management inline in the Monitor**: each PBS destination row exposes Download / Upload / Delete for the keyfile plus a Yes/No + passphrase + contextual Apply toggle for the escrow. When the installed keyfile does not match the backup's manifest, View contents / Download / Restore now show a structured amber panel with the required fingerprint so the user knows which keyfile to import.
---
@@ -61,7 +192,7 @@ For users who do **not** use an AI enhancement agent, the templated body now add
- **USB-NVMe / USB-SATA SMART on `removable=0` enclosures** — enclosures reporting `removable=0` (ASMedia, JMicron, Realtek, ASM105x) now walk sysfs to detect USB attachment, so `-d snt*` pass-through is tried and the drive's real model, serial, temperature, power-on hours and health surface. Temperature history sampler picks up the same fix.
- **PBS encryption prompt reworked** to a single explicit *Encrypt this backup?* Yes/No — nothing is uploaded to PBS unless the answer is Yes. Only when a keyfile is not yet installed does a second dialog ask whether to generate a new one or import an existing one. Cancelling never leaves a phantom keyfile behind.
- **Attached scheduled backups now inherit retention on every run** — jobs attached to a PVE vzdump parent re-read the parent's `prune-backups` config at each run and rewrite `KEEP_*` accordingly. Previously frozen to the value at job creation time.
- **Installer no longer auto-relaunches `menu` after an update** — the `exec MENU_SCRIPT` at the tail of the update path triggered *"line: syntax"* errors when bash tried to read the just-rewritten `/usr/local/bin/menu` under its feet. Flow now exits cleanly; operator types `menu` when ready. `change_release_channel` in Settings unaffected.
- **Installer no longer auto-relaunches `menu` after an update** — the `exec MENU_SCRIPT` at the tail of the update path triggered *"line: syntax"* errors when bash tried to read the just-rewritten `/usr/local/bin/menu` under its feet. Flow now exits cleanly; user types `menu` when ready. `change_release_channel` in Settings unaffected.
- **PBS restore listing broken on Proxmox 9 / jq 1.7**`hb_pbs_list_snapshots` switched from the prefix form `and not (...)` (rejected by jq 1.7) to the postfix form `and ((...) | not)` (accepted by both jq 1.6 and 1.7). Silent stderr redirect removed so future parse errors surface.
- **`run_scheduled_backup.sh` no longer crashes when `LANGUAGE` is unset** — cron / systemd invocations now load language + initialize the translation cache before sourcing utility functions that require it.
- **Local archive restore prompts no longer freeze silently**`hb_prompt_restore_source_dir` and `hb_prompt_local_archive` use the fd-9 TTY handoff already applied elsewhere.
@@ -108,7 +239,7 @@ Special thanks to the community members who shaped this release with concrete de
- **[@JF_Car](https://github.com/JF_Car)** — proposed the tree layout for the new Network Flow diagram so it reads correctly on mobile devices.
- **[@ghosthvj](https://github.com/ghosthvj)** — contributed the design for the new **Physical Disks** and **Physical Interfaces** cards.
- **[@riglesias](https://github.com/riglesias)**, **[@princo56](https://github.com/princo56)** and **[@jonatanc](https://github.com/jonatanc)** — tested the beta cycle end-to-end and provided the suggestions that closed most of the operator-visible gaps.
- **[@riglesias](https://github.com/riglesias)**, **[@princo56](https://github.com/princo56)** and **[@jonatanc](https://github.com/jonatanc)** — tested the beta cycle end-to-end and provided the suggestions that closed most of the user-visible gaps.
And to every user who opened an issue, commented in [GitHub Discussions](https://github.com/MacRimi/ProxMenux/discussions), reported a bug on the community channel, or told us what worked and what didn't on their hardware — most internal fixes in this release started as one of those reports. Keep them coming.
@@ -118,13 +249,13 @@ And to every user who opened an issue, commented in [GitHub Discussions](https:/
### New version ProxMenux v1.2.2 — *Stable consolidation of the v1.2.1.x cycle*
Stable release that brings the four prereleases of the **v1.2.1.x** cycle to the main channel in one move. The work over those four betas centred on three themes: making the Health Monitor genuinely configurable instead of just observable (per-category thresholds, per-event dismiss durations, an audit log of active suppressions), expanding the notification stack to cover roughly 80 services through Apprise while persisting events across Quiet Hours, and turning the Monitor process itself into a quieter, more predictable system citizen on idle hosts. On top of those, this release lands automatic upgrade detection for LXC containers, an end-to-end rewrite of the Coral TPU installer with the latest upstream drivers, and a long list of operator-visible fixes — HTTPS terminal handshake, kernel-update detection on PVE 9.x, NVIDIA installer flow on Alpine LXC, mixed-GPU passthrough audio companion handling, and several runtime optimizations on the Monitor scanning loops. Five direct code contributions from the community ship alongside ([@jcastro](https://github.com/jcastro) ×5, [@pespinel](https://github.com/pespinel) ×1) and the GPU passthrough work was driven by [@ghosthvj](https://github.com/ghosthvj)'s detailed field reports — see the Acknowledgments at the end.
Stable release that brings the four prereleases of the **v1.2.1.x** cycle to the main channel in one move. The work over those four betas centred on three themes: making the Health Monitor genuinely configurable instead of just observable (per-category thresholds, per-event dismiss durations, an audit log of active suppressions), expanding the notification stack to cover roughly 80 services through Apprise while persisting events across Quiet Hours, and turning the Monitor process itself into a quieter, more predictable system citizen on idle hosts. On top of those, this release lands automatic upgrade detection for LXC containers, an end-to-end rewrite of the Coral TPU installer with the latest upstream drivers, and a long list of user-visible fixes — HTTPS terminal handshake, kernel-update detection on PVE 9.x, NVIDIA installer flow on Alpine LXC, mixed-GPU passthrough audio companion handling, and several runtime optimizations on the Monitor scanning loops. Five direct code contributions from the community ship alongside ([@jcastro](https://github.com/jcastro) ×5, [@pespinel](https://github.com/pespinel) ×1) and the GPU passthrough work was driven by [@ghosthvj](https://github.com/ghosthvj)'s detailed field reports — see the Acknowledgments at the end.
---
## 🩺 Health Monitor — Configurable, Granular, Auditable
Three coupled pieces that together let the operator tune the Health Monitor to the actual envelope of their host instead of working around its defaults, and to manage dismisses with the same fine-grained control they already have over the rest of the dashboard.
Three coupled pieces that together let the user tune the Health Monitor to the actual envelope of their host instead of working around its defaults, and to manage dismisses with the same fine-grained control they already have over the rest of the dashboard.
### Per-category Warning / Critical thresholds
@@ -144,7 +275,7 @@ The *Dismiss* button on each Health Monitor alert now opens a small dropdown wit
- **7 days** — handy for a temporary condition you don't want to hear about during a week-long migration
- **Permanently** — silences this specific `error_key` indefinitely
Permanent dismisses persist with `suppression_hours = -1` in the persistence DB, never re-emit, never re-notify and are marked with a distinct amber **Permanent** badge in the Health Monitor so the operator always knows which alerts are intentionally silenced. The backend infrastructure for the permanent sentinel already existed — the UI just lacked a way to set it. The API contract is small and backwards-compatible: `POST /api/health/acknowledge` accepts an optional `suppression_hours` body field (positive integer for hours, `-1` for permanent); omitting it preserves the previous behaviour and uses the category's configured suppression. A second new endpoint `POST /api/health/un-acknowledge {error_key}` clears a previously-recorded acknowledgment so the alert becomes eligible to fire again — used by the Active Suppressions panel below.
Permanent dismisses persist with `suppression_hours = -1` in the persistence DB, never re-emit, never re-notify and are marked with a distinct amber **Permanent** badge in the Health Monitor so the user always knows which alerts are intentionally silenced. The backend infrastructure for the permanent sentinel already existed — the UI just lacked a way to set it. The API contract is small and backwards-compatible: `POST /api/health/acknowledge` accepts an optional `suppression_hours` body field (positive integer for hours, `-1` for permanent); omitting it preserves the previous behaviour and uses the category's configured suppression. A second new endpoint `POST /api/health/un-acknowledge {error_key}` clears a previously-recorded acknowledgment so the alert becomes eligible to fire again — used by the Active Suppressions panel below.
### Active Suppressions panel in Settings
@@ -172,7 +303,7 @@ Three reliability fixes ship alongside, all surfaced after the initial beta roll
2. **Backend whitelist regression** that rejected Apprise with HTTP 400. The notifications-test validator's hard-coded channel set (`{telegram, gotify, discord, email, all}`) was missing `apprise`, so every Apprise test or send returned `400 Invalid channel` before the library was even invoked. The whitelist is now derived live from `notification_channels.CHANNEL_TYPES`, so adding a new channel implementation in the future cannot silently regress this validator again.
3. **Opaque error reporting** when the destination returned a non-2xx response. When a destination (`jsons://`, `ntfy://`, `slack://`, …) rejected the payload, the operator only saw a generic *"Apprise rejected the notification (transport failure)"* message. The channel now captures Apprise's internal logger during `notify()` and surfaces the real HTTP status code plus the destination's response body (capped at 300 chars) — so a beta tester debugging a custom webhook can immediately see whether the upstream server is rejecting their payload schema.
3. **Opaque error reporting** when the destination returned a non-2xx response. When a destination (`jsons://`, `ntfy://`, `slack://`, …) rejected the payload, the user only saw a generic *"Apprise rejected the notification (transport failure)"* message. The channel now captures Apprise's internal logger during `notify()` and surfaces the real HTTP status code plus the destination's response body (capped at 300 chars) — so a beta tester debugging a custom webhook can immediately see whether the upstream server is rejecting their payload schema.
---
@@ -209,7 +340,7 @@ The mount monitor used to call `lxc-info -n <vmid> -p` for every running CT just
## 🔌 HTTPS Terminal Handshake
Every terminal modal in the Monitor (dashboard terminal, LXC terminal, script terminal) used to fail with *WebSocket connection error* on hosts where HTTPS was enabled. The root cause was specific to the `gevent + SSL` path: the gevent-websocket `WebSocketHandler` was stacked on top of flask-sock's protocol implementation, so the server emitted **two** consecutive `HTTP/1.1 101 Switching Protocols` headers and the browser closed the connection as a corrupt frame. Dropping the explicit `handler_class=WebSocketHandler` argument restores a single 101 response and the handshake completes normally. The fix is invisible to operators running on plain HTTP — they were unaffected — but unblocks every HTTPS-fronted install (reverse proxies, certificate-managed deployments, anything behind nginx/Traefik).
Every terminal modal in the Monitor (dashboard terminal, LXC terminal, script terminal) used to fail with *WebSocket connection error* on hosts where HTTPS was enabled. The root cause was specific to the `gevent + SSL` path: the gevent-websocket `WebSocketHandler` was stacked on top of flask-sock's protocol implementation, so the server emitted **two** consecutive `HTTP/1.1 101 Switching Protocols` headers and the browser closed the connection as a corrupt frame. Dropping the explicit `handler_class=WebSocketHandler` argument restores a single 101 response and the handshake completes normally. The fix is invisible to users running on plain HTTP — they were unaffected — but unblocks every HTTPS-fronted install (reverse proxies, certificate-managed deployments, anything behind nginx/Traefik).
Additionally, the terminal panel used to lose its WebSocket connection when the user enabled the browser's auto-translate feature (Chrome / Edge / Safari "translate this page" prompts). The translator moves DOM nodes that React still holds refs to, and the WebSocket React component breaks because its container ref points to a moved node. Added `translate="no"` on the terminal container divs so the translator skips the embedded tty entirely — translations on the rest of the page still work.
@@ -223,7 +354,7 @@ On Proxmox VE 9.x hosts, the *System Updates → Kernel / PVE* row used to repor
2. **Dry-run switched from `apt-get upgrade --dry-run` to `apt-get dist-upgrade --dry-run`**. PVE 9 ships kernel updates packaged as new installs (not as straight upgrades of an existing package), and the plain `upgrade --dry-run` does not consider new installs at all. `dist-upgrade --dry-run` does.
3. **Running-kernel detection** now reads `uname -r` and flags an update as a *running-kernel update* when the package matches the running release exactly or its branch meta-package (e.g. `proxmox-kernel-6.14` for a host on `6.14.11-4-pve`). The row text distinguishes *"Running kernel update available (reboot required)"* from *"N kernel update(s) available (none for running kernel)"* so the operator knows whether they need to reboot or just install.
3. **Running-kernel detection** now reads `uname -r` and flags an update as a *running-kernel update* when the package matches the running release exactly or its branch meta-package (e.g. `proxmox-kernel-6.14` for a host on `6.14.11-4-pve`). The row text distinguishes *"Running kernel update available (reboot required)"* from *"N kernel update(s) available (none for running kernel)"* so the user knows whether they need to reboot or just install.
---
@@ -253,9 +384,9 @@ New documentation pages cover the **Active Suppressions** section in the Setting
- **Post-install function update detection** — the Monitor tracks installed ProxMenux optimizations (Log2Ram, Memory Settings, System Limits, Logrotate, …) and notifies when a newer version is available, with one-click apply from Settings.
- **Secure Gateway (Tailscale) update flow** — one-click Tailscale update from Settings with Last-checked / Installed / Latest indicators and notification when a new version is published.
- **Helper-Scripts menu** — richer context and useful information for each entry, making it easier to know what every script does before running it.
- **Burst aggregation wording** — burst summaries now report only the *additional* events that arrived after the initial individual alert, so the operator no longer sees the first event counted twice.
- **Burst aggregation wording** — burst summaries now report only the *additional* events that arrived after the initial individual alert, so the user no longer sees the first event counted twice.
- **Known-error classifier** — word-boundary regex on ATA / UNC patterns so kernel messages like `nvidia_uvm:FatalError` are no longer misclassified as ATA cable issues.
- **VM / CT control errors** — failed start / stop / restart now surfaces the real `pvesh` stderr (e.g. *"no space left on device"*) in the UI toast and fires a `vm_fail` / `ct_fail` notification, instead of the bare 500 INTERNAL SERVER ERROR the operator used to see.
- **VM / CT control errors** — failed start / stop / restart now surfaces the real `pvesh` stderr (e.g. *"no space left on device"*) in the UI toast and fires a `vm_fail` / `ct_fail` notification, instead of the bare 500 INTERNAL SERVER ERROR the user used to see.
- **log2ram apply path** — the auto / update flow now restarts log2ram after writing the new size, so a configured `512M` actually takes effect on the running tmpfs without a manual restart.
- **PVE webhook URL** — the notification webhook now follows the active SSL state automatically, switching between `http://` and `https://` when you toggle HTTPS in the panel.
- **Frontend 401 cascade** — the login screen no longer swallows a 401 forever after a brief stale-token state; the dedup flag is cleared on mount and on successful login.
@@ -1352,4 +1483,4 @@ Disks now display tags like ⚠ In use, ⚠ RAID, ⚠ LVM, or ⚠ ZFS, making it
## [1.0.0] - 2024-12-18
### Added
- Initial release of **ProxMenux**.
- Created a script to add **Coral TPU drivers** to Proxmox.
- Created a script to add **Coral TPU drivers** to Proxmox.
+14 -9
View File
@@ -32,7 +32,7 @@
<br />
<p align="center">
<strong>ProxMenux</strong> is a management tool for <strong>Proxmox VE</strong> that simplifies system administration through an interactive menu, allowing you to execute commands and scripts with ease.
<strong>ProxMenux</strong> is an interactive toolkit for <strong>Proxmox VE</strong> — a menu-driven CLI plus a web dashboard for post-install, backup/restore and continuous health monitoring of your homelab.
</p>
---
@@ -138,7 +138,7 @@ journalctl -u proxmenux-monitor -n 50
## 🔧 Dependencies
The following dependencies are installed automatically during setup:
The following Debian packages are installed automatically during setup:
| Package | Purpose |
|---|---|
@@ -146,12 +146,9 @@ The following dependencies are installed automatically during setup:
| `curl` | Downloads and connectivity checks |
| `jq` | JSON processing |
| `git` | Repository cloning and updates |
| `python3` + `python3-venv` | Translation support *(Translation version only)* |
| `googletrans` | Google Translate library *(Translation version only)* |
| `python3` + `python3-pip` | ProxMenux Monitor (Flask web dashboard) |
> **🛡️ Security Note / VirusTotal False Positive**
>
> If you scan the raw installation URL on VirusTotal, you might see a 1/95 detection by heuristic engines like *Chong Lua Dao*. This is a **known false positive**. Because this script uses the standard `curl | bash` installation pattern and downloads legitimate binaries (like `jq` from its official GitHub release), overly aggressive scanners flag the *behavior*. The script is 100% open source and safe to review. You can read more about this in [Issue #162](https://github.com/MacRimi/ProxMenux/issues/162).
UI translations ship as pre-built JSON files per language (English, Spanish, French, German, Italian, Portuguese). There is no runtime translation dependency.
---
@@ -198,6 +195,14 @@ If you want to go a step further, a coffee on Ko-fi keeps development going:
---
## 📈 Star History
## 📈 Project Growth
[![Star History Chart](https://api.star-history.com/svg?repos=MacRimi/ProxMenux&type=Date)](https://www.star-history.com/#MacRimi/ProxMenux&Date)
<p align="center">
<a href="https://github.com/MacRimi/repo-growth">
<img src="images/project-growth.svg" alt="ProxMenux project growth" width="900">
</a>
</p>
<p align="center">
<sub>Stars and forks tracked with <a href="https://github.com/MacRimi/repo-growth">Repo Growth</a>.</sub>
</p>
+1 -1
View File
@@ -1 +1 @@
1.2.1.4
1.2.4.0
File diff suppressed because one or more lines are too long

After

Width:  |  Height:  |  Size: 25 KiB

+115 -406
View File
@@ -44,11 +44,15 @@ LOCAL_SCRIPTS="/usr/local/share/proxmenux/scripts"
INSTALL_DIR="/usr/local/bin"
BASE_DIR="/usr/local/share/proxmenux"
CONFIG_FILE="$BASE_DIR/config.json"
CACHE_FILE="$BASE_DIR/cache.json"
UTILS_FILE="$BASE_DIR/utils.sh"
LOCAL_VERSION_FILE="$BASE_DIR/version.txt"
MENU_SCRIPT="menu"
VENV_PATH="/opt/googletrans-env"
# Legacy path that existed during the Python+googletrans era. The current
# translate flow uses pre-generated JSON files in lang/ — no virtualenv,
# no online translation at runtime — so this path is purged on install
# if present. Kept as a literal here so the cleanup is grep-able.
LEGACY_VENV_PATH="/opt/googletrans-env"
MONITOR_INSTALL_DIR="$BASE_DIR"
MONITOR_RUNTIME_DIR="$BASE_DIR/monitor-app"
@@ -272,10 +276,6 @@ cleanup_corrupted_files() {
echo "Cleaning up corrupted configuration file..."
rm -f "$CONFIG_FILE"
fi
if [ -f "$CACHE_FILE" ] && ! jq empty "$CACHE_FILE" >/dev/null 2>&1; then
echo "Cleaning up corrupted cache file..."
rm -f "$CACHE_FILE"
fi
}
# Cleanup function
@@ -291,157 +291,27 @@ trap cleanup EXIT
# ==========================================================
check_existing_installation() {
local has_venv=false
local has_config=false
local has_language=false
local has_menu=false
# After the googletrans removal there is only one install variant.
# The function still distinguishes "installed" vs "not installed" so
# show_installation_options can pick the right banner.
if [ -f "$INSTALL_DIR/$MENU_SCRIPT" ]; then
has_menu=true
fi
if [ -d "$VENV_PATH" ] && [ -f "$VENV_PATH/bin/activate" ]; then
has_venv=true
fi
if [ -f "$CONFIG_FILE" ]; then
if jq empty "$CONFIG_FILE" >/dev/null 2>&1; then
has_config=true
local current_language=$(jq -r '.language // empty' "$CONFIG_FILE" 2>/dev/null)
if [[ -n "$current_language" && "$current_language" != "null" && "$current_language" != "empty" ]]; then
has_language=true
fi
else
echo "Warning: Corrupted config file detected, removing..."
# Quietly fix a corrupted config so the install can proceed.
if [ -f "$CONFIG_FILE" ] && ! jq empty "$CONFIG_FILE" >/dev/null 2>&1; then
echo "Warning: Corrupted config file detected, removing..." >&2
rm -f "$CONFIG_FILE"
fi
fi
if [ "$has_venv" = true ] && [ "$has_language" = true ]; then
echo "translation"
elif [ "$has_menu" = true ] && [ "$has_venv" = false ]; then
echo "normal"
elif [ "$has_menu" = true ]; then
echo "unknown"
echo "installed"
else
echo "none"
fi
}
uninstall_proxmenux() {
local install_type="$1"
local force_clean="$2"
if [ "$force_clean" != "force" ]; then
if ! whiptail --title "Uninstall ProxMenux" --yesno "Are you sure you want to uninstall ProxMenux?" 10 60; then
return 1
fi
fi
echo "Uninstalling ProxMenux..."
if systemctl is-active --quiet proxmenux-monitor.service; then
echo "Stopping ProxMenux Monitor service..."
systemctl stop proxmenux-monitor.service
fi
if systemctl is-enabled --quiet proxmenux-monitor.service 2>/dev/null; then
echo "Disabling ProxMenux Monitor service..."
systemctl disable proxmenux-monitor.service
fi
if [ -f "$MONITOR_SERVICE_FILE" ]; then
echo "Removing ProxMenux Monitor service file..."
rm -f "$MONITOR_SERVICE_FILE"
systemctl daemon-reload
fi
if [ -d "$MONITOR_INSTALL_DIR" ]; then
echo "Removing ProxMenux Monitor directory..."
rm -rf "$MONITOR_INSTALL_DIR"
fi
if [ -f "$VENV_PATH/bin/activate" ]; then
echo "Removing googletrans and virtual environment..."
source "$VENV_PATH/bin/activate"
pip uninstall -y googletrans >/dev/null 2>&1
deactivate
rm -rf "$VENV_PATH"
fi
if [ "$install_type" = "translation" ] && [ "$force_clean" != "force" ]; then
DEPS_TO_REMOVE=$(whiptail --title "Remove Translation Dependencies" --checklist \
"Select translation-specific dependencies to remove:" 15 60 3 \
"python3-venv" "Python virtual environment" OFF \
"python3-pip" "Python package installer" OFF \
"python3" "Python interpreter" OFF \
3>&1 1>&2 2>&3)
if [ -n "$DEPS_TO_REMOVE" ]; then
echo "Removing selected dependencies..."
read -r -a DEPS_ARRAY <<< "$(echo "$DEPS_TO_REMOVE" | tr -d '"')"
for dep in "${DEPS_ARRAY[@]}"; do
echo "Removing $dep..."
apt-mark auto "$dep" >/dev/null 2>&1
apt-get -y --purge autoremove "$dep" >/dev/null 2>&1
done
apt-get autoremove -y --purge >/dev/null 2>&1
fi
fi
rm -f "$INSTALL_DIR/$MENU_SCRIPT"
rm -rf "$BASE_DIR"
[ -f /root/.bashrc.bak ] && mv /root/.bashrc.bak /root/.bashrc
if [ -f /etc/motd.bak ]; then
mv /etc/motd.bak /etc/motd
else
sed -i '/This system is optimised by: ProxMenux/d' /etc/motd
fi
echo "ProxMenux has been uninstalled."
return 0
}
handle_installation_change() {
local current_type="$1"
local new_type="$2"
if [ "$current_type" = "$new_type" ]; then
return 0
fi
case "$current_type-$new_type" in
"translation-1"|"translation-normal")
if whiptail --title "Installation Type Change" \
--yesno "Switch from Translation to Normal Version?\n\nThis will remove translation components." 10 60; then
echo "Preparing for installation type change..."
uninstall_proxmenux "translation" "force" >/dev/null 2>&1
return 0
else
return 1
fi
;;
"normal-2"|"normal-translation")
if whiptail --title "Installation Type Change" \
--yesno "Switch from Normal to Translation Version?\n\nThis will add translation components." 10 60; then
return 0
else
return 1
fi
;;
*)
return 0
;;
esac
}
update_config() {
local component="$1"
local status="$2"
local timestamp=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
local tracked_components=("dialog" "curl" "jq" "git" "python3" "python3-venv" "python3-pip" "virtual_environment" "pip" "googletrans" "proxmenux_monitor")
local tracked_components=("dialog" "curl" "jq" "git" "python3" "python3-pip" "proxmenux_monitor")
if [[ " ${tracked_components[@]} " =~ " ${component} " ]]; then
mkdir -p "$(dirname "$CONFIG_FILE")"
@@ -485,7 +355,7 @@ select_language() {
fi
LANGUAGE=$(whiptail --title "Select Language" --menu "Choose a language for the menu:" 20 60 12 \
"en" "English (Recommended)" \
"en" "English" \
"es" "Spanish" \
"fr" "French" \
"de" "German" \
@@ -517,26 +387,12 @@ select_language() {
# Show installation confirmation for new installations
show_installation_confirmation() {
local install_type="$1"
case "$install_type" in
"1")
if whiptail --title "ProxMenux - Normal Version Installation" \
--yesno "ProxMenux Normal Version will install:\n\n• dialog (interactive menus) - Official Debian package\n• curl (file downloads) - Official Debian package\n• jq (JSON processing) - Official Debian package\n• ProxMenux core files (/usr/local/share/proxmenux)\n• ProxMenux Monitor (Web dashboard on port 8008)\n\nThis is a lightweight installation with minimal dependencies.\n\nProceed with installation?" 20 70; then
return 0
else
return 1
fi
;;
"2")
if whiptail --title "ProxMenux - Translation Version Installation" \
--yesno "ProxMenux Translation Version will install:\n\n• dialog (interactive menus)\n• curl (file downloads)\n• jq (JSON processing)\n• python3 + python3-venv + python3-pip\n• Google Translate library (googletrans)\n• Virtual environment (/opt/googletrans-env)\n• Translation cache system\n• ProxMenux core files\n• ProxMenux Monitor (Web dashboard on port 8008)\n\nThis version requires more dependencies for translation support.\n\nProceed with installation?" 20 70; then
return 0
else
return 1
fi
;;
esac
if whiptail --title "ProxMenux Installation" \
--yesno "ProxMenux will install:\n\n• dialog (interactive menus) - Official Debian package\n• curl (file downloads) - Official Debian package\n• jq (JSON processing) - Official Debian package\n• ProxMenux core files (/usr/local/share/proxmenux)\n• ProxMenux Monitor (Web dashboard on port 8008)\n• Pre-built translation files\n\nProceed with installation?" 20 70; then
return 0
else
return 1
fi
}
get_server_ip() {
@@ -629,7 +485,7 @@ extract_appimage_to_runtime_dir() {
rm -f "$appimage_path"
msg_ok "AppImage runtime extracted (no FUSE mount; bypasses Wazuh rule 521)."
msg_ok "AppImage runtime extracted."
return 0
}
@@ -800,14 +656,33 @@ EOF
}
install_normal_version() {
local total_steps=5
local total_steps=6
local current_step=1
# Translations now live as pre-generated JSON files under lang/, so
# asking the language up front is the right place — every install is
# multilingual-capable and the user picks once.
show_progress $current_step $total_steps "Language selection"
select_language
((current_step++))
# Purge the legacy googletrans virtualenv if a previous install left it
# behind. The new translate flow has no runtime Python/googletrans
# dependency — the venv is dead weight on disk now.
if [[ -d "$LEGACY_VENV_PATH" ]]; then
msg_info "Removing legacy translation virtualenv at $LEGACY_VENV_PATH..."
rm -rf "$LEGACY_VENV_PATH"
msg_ok "Legacy translation virtualenv removed."
fi
show_progress $current_step $total_steps "Installing basic dependencies."
msg_info "Refreshing apt cache..."
apt-get update -y > /dev/null 2>&1 || true
msg_ok "apt cache refreshed."
msg_info "Installing jq..."
if ! command -v jq > /dev/null 2>&1; then
apt-get update > /dev/null 2>&1
if apt-get install -y jq > /dev/null 2>&1 && command -v jq > /dev/null 2>&1; then
update_config "jq" "installed"
else
@@ -829,20 +704,14 @@ install_normal_version() {
else
update_config "jq" "already_installed"
fi
msg_ok "jq ready."
BASIC_DEPS=("dialog" "curl" "git")
if [ -z "${APT_UPDATED:-}" ]; then
apt-get update -y > /dev/null 2>&1 || true
APT_UPDATED=1
fi
for pkg in "${BASIC_DEPS[@]}"; do
# Strict per-package check — see comment in install_translation_version().
msg_info "Installing $pkg..."
# dpkg-query for the EXACT package — `dpkg -l | grep -qw python3`
# falsely matches `python3-pip`. Issue #205.
if ! dpkg-query -W -f='${Status}' "$pkg" 2>/dev/null | grep -q "ok installed"; then
if apt-get install -y "$pkg" > /dev/null 2>&1; then
update_config "$pkg" "installed"
@@ -854,6 +723,7 @@ install_normal_version() {
else
update_config "$pkg" "already_installed"
fi
msg_ok "$pkg ready."
done
@@ -894,7 +764,7 @@ install_normal_version() {
((current_step++))
show_progress $current_step $total_steps "Install ProxMenux repository"
msg_info "Cloning ProxMenux repositoryy."
msg_info "Cloning ProxMenux repository."
if ! git clone --depth 1 "$REPO_URL" "$TEMP_DIR" 2>/dev/null; then
msg_error "Failed to clone repository from $REPO_URL"
exit 1
@@ -921,10 +791,28 @@ install_normal_version() {
show_progress $current_step $total_steps "Copying necessary files"
cp "./scripts/utils.sh" "$UTILS_FILE"
cp "./menu" "$INSTALL_DIR/$MENU_SCRIPT"
# Atomic install of /usr/local/bin/menu: stage to .new on the same
# filesystem then mv. This protects any reader that happens to open
# the file mid-install from seeing a partial/half-written script
# (the suspected root cause of the post-1.2.2-update reports:
# "menu: line 138 syntax error near unexpected token `$REMOTE_VERSION`")
cp "./menu" "$INSTALL_DIR/${MENU_SCRIPT}.new"
mv -f "$INSTALL_DIR/${MENU_SCRIPT}.new" "$INSTALL_DIR/$MENU_SCRIPT"
cp "./version.txt" "$LOCAL_VERSION_FILE"
cp "./install_proxmenux.sh" "$BASE_DIR/install_proxmenux.sh"
# Pre-built translation cache. The runtime translate() in utils.sh
# reads $BASE_DIR/lang/<lang>.json — these files ship with the repo
# (one per supported language) so the install ends up multilingual
# without any runtime download or Python dependency. Refresh the
# whole dir on every install so a language that was renamed or
# dropped upstream disappears here too.
if [ -d "./lang" ]; then
rm -rf "$BASE_DIR/lang"
mkdir -p "$BASE_DIR/lang"
cp -r "./lang/"* "$BASE_DIR/lang/" 2>/dev/null || true
fi
# A user that previously rode the beta train and then switched back
# to stable would still have a leftover beta_version.txt under
# $BASE_DIR, which makes the `menu` update check (check_updates_beta)
@@ -936,7 +824,7 @@ install_normal_version() {
# Wipe the scripts tree before copying so any file removed upstream
# (renamed, consolidated, deprecated) disappears from the user install.
# Only $BASE_DIR/scripts/ is cleared; config.json, cache.json,
# Only $BASE_DIR/scripts/ is cleared; config.json,
# components_status.json, version.txt, monitor.db, smart/, oci/ and
# the AppImage live outside this path and are preserved.
rm -rf "$BASE_DIR/scripts"
@@ -960,207 +848,19 @@ install_normal_version() {
create_monitor_service
fi
msg_ok "ProxMenux Normal Version installation completed successfully."
}
install_translation_version() {
local total_steps=5
local current_step=1
show_progress $current_step $total_steps "Language selection"
select_language
((current_step++))
show_progress $current_step $total_steps "Installing system dependencies"
if ! command -v jq > /dev/null 2>&1; then
apt-get update > /dev/null 2>&1
if apt-get install -y jq > /dev/null 2>&1 && command -v jq > /dev/null 2>&1; then
update_config "jq" "installed"
else
local jq_url="https://github.com/jqlang/jq/releases/download/jq-1.7.1/jq-linux-amd64"
if wget -q -O /usr/local/bin/jq "$jq_url" 2>/dev/null && chmod +x /usr/local/bin/jq; then
if command -v jq > /dev/null 2>&1; then
update_config "jq" "installed_from_github"
else
msg_error "Failed to install jq. Please install it manually."
update_config "jq" "failed"
return 1
fi
else
msg_error "Failed to install jq from both APT and GitHub. Please install it manually."
update_config "jq" "failed"
return 1
fi
fi
else
update_config "jq" "already_installed"
fi
DEPS=("dialog" "curl" "git" "python3" "python3-venv" "python3-pip")
for pkg in "${DEPS[@]}"; do
# `dpkg -l | grep -qw "$pkg"` treats `-` as a word boundary, so a
# query for `python3` would falsely match `python3-pip` and skip
# the real `python3` install. `dpkg-query -W -f='${Status}'` asks
# for the EXACT package and reports "install ok installed" only
# when truly present. Issue #205 traced back here.
if ! dpkg-query -W -f='${Status}' "$pkg" 2>/dev/null | grep -q "ok installed"; then
if apt-get install -y "$pkg" > /dev/null 2>&1; then
update_config "$pkg" "installed"
else
msg_error "Failed to install $pkg. Please install it manually."
update_config "$pkg" "failed"
return 1
fi
else
update_config "$pkg" "already_installed"
fi
done
msg_ok "jq, dialog, curl, git, python3, python3-venv and python3-pip installed successfully."
((current_step++))
show_progress $current_step $total_steps "Setting up translation environment"
if [ ! -d "$VENV_PATH" ] || [ ! -f "$VENV_PATH/bin/activate" ]; then
python3 -m venv --system-site-packages "$VENV_PATH" > /dev/null 2>&1
if [ ! -f "$VENV_PATH/bin/activate" ]; then
msg_error "Failed to create virtual environment. Please check your Python installation."
update_config "virtual_environment" "failed"
return 1
else
update_config "virtual_environment" "created"
fi
else
update_config "virtual_environment" "already_exists"
fi
source "$VENV_PATH/bin/activate"
if pip install --upgrade pip > /dev/null 2>&1; then
update_config "pip" "upgraded"
else
msg_error "Failed to upgrade pip."
update_config "pip" "upgrade_failed"
return 1
fi
if pip install --break-system-packages --no-cache-dir googletrans==4.0.0-rc1 > /dev/null 2>&1; then
update_config "googletrans" "installed"
else
msg_error "Failed to install googletrans. Please check your internet connection."
update_config "googletrans" "failed"
deactivate
return 1
fi
deactivate
show_progress $current_step $total_steps "Cloning ProxMenux repository"
if ! git clone --depth 1 "$REPO_URL" "$TEMP_DIR" 2>/dev/null; then
msg_error "Failed to clone repository from $REPO_URL"
exit 1
fi
msg_ok "Repository cloned successfully."
cd "$TEMP_DIR"
((current_step++))
show_progress $current_step $total_steps "Copying necessary files"
mkdir -p "$BASE_DIR"
mkdir -p "$INSTALL_DIR"
cp "./json/cache.json" "$CACHE_FILE"
msg_ok "Cache file copied with translations."
cp "./scripts/utils.sh" "$UTILS_FILE"
cp "./menu" "$INSTALL_DIR/$MENU_SCRIPT"
cp "./version.txt" "$LOCAL_VERSION_FILE"
cp "./install_proxmenux.sh" "$BASE_DIR/install_proxmenux.sh"
# Clear any leftover beta_version.txt — see the equivalent block
# in the update path above for the rationale (in short: prevents
# the menu from offering a phantom "Beta update available" after a
# user has switched back to the stable channel).
rm -f "$BASE_DIR/beta_version.txt"
mkdir -p "$BASE_DIR/scripts"
cp -r "./scripts/"* "$BASE_DIR/scripts/"
chmod -R +x "$BASE_DIR/scripts/"
chmod +x "$BASE_DIR/install_proxmenux.sh"
msg_ok "Necessary files created."
chmod +x "$INSTALL_DIR/$MENU_SCRIPT"
((current_step++))
show_progress $current_step $total_steps "Installing ProxMenux Monitor"
install_proxmenux_monitor
local monitor_status=$?
if [ $monitor_status -eq 0 ]; then
create_monitor_service
elif [ $monitor_status -eq 2 ]; then
msg_ok "ProxMenux Monitor updated successfully."
fi
msg_ok "ProxMenux Translation Version installation completed successfully."
msg_ok "ProxMenux installation completed successfully."
}
show_installation_options() {
local current_install_type
current_install_type=$(check_existing_installation)
local pve_version
pve_version=$(pveversion 2>/dev/null | grep -oP 'pve-manager/\K[0-9]+' | head -1)
local menu_title="ProxMenux Installation"
local menu_text="Choose installation type:"
if [ "$current_install_type" != "none" ]; then
case "$current_install_type" in
"translation")
menu_title="ProxMenux Update - Translation Version Detected"
;;
"normal")
menu_title="ProxMenux Update - Normal Version Detected"
;;
"unknown")
menu_title="ProxMenux Update - Existing Installation Detected"
;;
esac
fi
if [[ "$pve_version" -ge 9 ]]; then
INSTALL_TYPE=$(whiptail --backtitle "ProxMenux" --title "$menu_title" --menu "\n$menu_text" 14 70 2 \
"1" "Normal Version (English only)" 3>&1 1>&2 2>&3)
if [ -z "$INSTALL_TYPE" ]; then
show_proxmenux_logo
msg_warn "Installation cancelled."
exit 1
fi
else
INSTALL_TYPE=$(whiptail --backtitle "ProxMenux" --title "$menu_title" --menu "\n$menu_text" 14 70 2 \
"1" "Normal Version (English only)" \
"2" "Translation Version (Multi-language support)" 3>&1 1>&2 2>&3)
if [ -z "$INSTALL_TYPE" ]; then
show_proxmenux_logo
msg_warn "Installation cancelled."
exit 1
fi
fi
if [ -z "$INSTALL_TYPE" ]; then
show_proxmenux_logo
msg_warn "Installation cancelled."
exit 1
fi
# Translation Version is gone — translations now ship as pre-built
# JSON files in lang/. There is only one install path, so this
# function just shows the confirmation dialog for new installs and
# then returns. Existing installs go straight through (they already
# consented to update via the menu).
INSTALL_TYPE="1"
if [ "$current_install_type" = "none" ]; then
if ! show_installation_confirmation "$INSTALL_TYPE"; then
show_proxmenux_logo
@@ -1168,33 +868,23 @@ show_installation_options() {
exit 1
fi
fi
if ! handle_installation_change "$current_install_type" "$INSTALL_TYPE"; then
show_proxmenux_logo
msg_warn "Installation cancelled."
exit 1
fi
}
install_proxmenux() {
show_installation_options
case "$INSTALL_TYPE" in
"1")
show_proxmenux_logo
msg_title "Installing ProxMenux - Normal Version"
install_normal_version
;;
"2")
show_proxmenux_logo
msg_title "Installing ProxMenux - Translation Version"
install_translation_version
;;
*)
msg_error "Invalid option selected."
exit 1
;;
esac
if [[ "${UPDATE_MODE:-0}" == "1" ]]; then
# Update path: the user already accepted "Update now?" in the
# menu. Hand off to the freshly-installed menu binary at the end
# (exec, see below) so no shell ever returns to a half-written
# /usr/local/bin/menu — the new copy is the only thing parsed.
show_proxmenux_logo
msg_title "Updating ProxMenux"
install_normal_version
else
show_installation_options
show_proxmenux_logo
msg_title "Installing ProxMenux"
install_normal_version
fi
if [[ -f "$UTILS_FILE" ]]; then
source "$UTILS_FILE"
@@ -1210,14 +900,27 @@ install_proxmenux() {
bash "$LOCAL_SCRIPTS/global/cleanup_gpu_hookscripts.sh" || true
fi
# UPDATE_MODE used to `exec "$INSTALL_DIR/$MENU_SCRIPT"` here to
# auto-relaunch the freshly-installed menu. That produced visible
# "line: syntax" errors when bash tried to read the just-rewritten
# /usr/local/bin/menu (or a shared utils.sh sourced by it) under
# its feet. Fall through to the standard "installed successfully"
# end-of-run message instead — the operator types `menu` when
# ready and the terminal is stable by then.
#
# `change_release_channel` in scripts/menus/config_menu.sh is
# unaffected: it invokes the installer without `--update`
# (UPDATE_MODE=0) so it never went through this branch, and its
# "return to config menu" behaviour is preserved.
msg_title "ProxMenux has been installed successfully"
if systemctl is-active --quiet proxmenux-monitor.service; then
local server_ip=$(get_server_ip)
echo -e "${GN}🌐 ProxMenux Monitor activated${CL}: ${BL}http://${server_ip}:${MONITOR_PORT}${CL}"
echo
fi
echo -ne "${GN}"
type_text "To run ProxMenux, simply execute this command in the console or terminal:"
echo -e "${YWB} menu${CL}"
@@ -1226,6 +929,12 @@ install_proxmenux() {
exit 0
}
# Parse CLI flags before anything else so install_proxmenux() can
# branch on UPDATE_MODE without re-reading "$@".
if [[ "${1:-}" == "--update" ]]; then
UPDATE_MODE=1
fi
if [ "$(id -u)" -ne 0 ]; then
msg_error "This script must be run as root."
exit 1
+125 -26
View File
@@ -35,12 +35,16 @@
INSTALL_DIR="/usr/local/bin"
BASE_DIR="/usr/local/share/proxmenux"
CONFIG_FILE="$BASE_DIR/config.json"
CACHE_FILE="$BASE_DIR/cache.json"
UTILS_FILE="$BASE_DIR/utils.sh"
LOCAL_VERSION_FILE="$BASE_DIR/version.txt"
BETA_VERSION_FILE="$BASE_DIR/beta_version.txt"
MENU_SCRIPT="menu"
# Legacy path that existed during the Python+googletrans era. Purged on
# install if present — the current translate flow uses pre-built JSON
# files in lang/ and has no runtime venv dependency.
LEGACY_VENV_PATH="/opt/googletrans-env"
MONITOR_INSTALL_DIR="$BASE_DIR"
MONITOR_RUNTIME_DIR="$BASE_DIR/monitor-app"
MONITOR_SERVICE_FILE="/etc/systemd/system/proxmenux-monitor.service"
@@ -304,9 +308,6 @@ cleanup_corrupted_files() {
if [ -f "$CONFIG_FILE" ] && ! jq empty "$CONFIG_FILE" >/dev/null 2>&1; then
rm -f "$CONFIG_FILE"
fi
if [ -f "$CACHE_FILE" ] && ! jq empty "$CACHE_FILE" >/dev/null 2>&1; then
rm -f "$CACHE_FILE"
fi
}
detect_latest_appimage() {
@@ -371,7 +372,7 @@ extract_appimage_to_runtime_dir() {
rm -f "$appimage_path"
msg_ok "AppImage runtime extracted (no FUSE mount; bypasses Wazuh rule 521)."
msg_ok "AppImage runtime extracted."
return 0
}
@@ -541,15 +542,78 @@ EOF
}
# ── Main install ───────────────────────────────────────────
select_language() {
if [ -f "$CONFIG_FILE" ] && jq empty "$CONFIG_FILE" >/dev/null 2>&1; then
local existing_language=$(jq -r '.language // empty' "$CONFIG_FILE" 2>/dev/null)
if [[ -n "$existing_language" && "$existing_language" != "null" && "$existing_language" != "empty" ]]; then
LANGUAGE="$existing_language"
msg_ok "Using existing language configuration: $LANGUAGE"
return 0
fi
fi
LANGUAGE=$(whiptail --title "Select Language" --menu "Choose a language for the menu:" 20 60 12 \
"en" "English" \
"es" "Spanish" \
"fr" "French" \
"de" "German" \
"it" "Italian" \
"pt" "Portuguese" 3>&1 1>&2 2>&3)
if [ -z "$LANGUAGE" ]; then
msg_error "No language selected. Exiting."
exit 1
fi
mkdir -p "$(dirname "$CONFIG_FILE")"
if [ ! -f "$CONFIG_FILE" ] || ! jq empty "$CONFIG_FILE" >/dev/null 2>&1; then
echo '{}' > "$CONFIG_FILE"
fi
local tmp_file
tmp_file=$(mktemp)
if jq --arg lang "$LANGUAGE" '. + {language: $lang}' "$CONFIG_FILE" > "$tmp_file" 2>/dev/null; then
mv "$tmp_file" "$CONFIG_FILE"
else
echo "{\"language\": \"$LANGUAGE\"}" > "$CONFIG_FILE"
fi
[ -f "$tmp_file" ] && rm -f "$tmp_file"
msg_ok "Language set to: $LANGUAGE"
}
install_beta() {
local total_steps=4
local total_steps=5
local current_step=1
# ── Step 1: Dependencies ──────────────────────────────
# ── Step 1: Language selection ────────────────────────
# Pre-built translations in lang/<locale>.json make every beta install
# multilingual-capable. Ask the operator once up front; subsequent
# update runs reuse the saved choice without re-prompting.
show_progress $current_step $total_steps "Language selection"
select_language
((current_step++))
# Purge the legacy googletrans virtualenv if a previous install left it
# behind. Runtime translation is now a static JSON lookup — the venv
# is dead weight on disk now.
if [[ -d "$LEGACY_VENV_PATH" ]]; then
msg_info "Removing legacy translation virtualenv at $LEGACY_VENV_PATH..."
rm -rf "$LEGACY_VENV_PATH"
msg_ok "Legacy translation virtualenv removed."
fi
# ── Step 2: Dependencies ──────────────────────────────
show_progress $current_step $total_steps "Installing system dependencies"
msg_info "Refreshing apt cache..."
apt-get update -y > /dev/null 2>&1 || true
msg_ok "apt cache refreshed."
msg_info "Installing jq..."
if ! command -v jq > /dev/null 2>&1; then
apt-get update > /dev/null 2>&1
if apt-get install -y jq > /dev/null 2>&1 && command -v jq > /dev/null 2>&1; then
update_config "jq" "installed"
else
@@ -566,18 +630,13 @@ install_beta() {
else
update_config "jq" "already_installed"
fi
msg_ok "jq ready."
local BASIC_DEPS=("dialog" "curl" "git")
if [ -z "${APT_UPDATED:-}" ]; then
apt-get update -y > /dev/null 2>&1 || true
APT_UPDATED=1
fi
for pkg in "${BASIC_DEPS[@]}"; do
# Strict per-package check — `dpkg -l | grep -qw python3` falsely
# matches `python3-pip` (the `-` is a word boundary), so dpkg-query
# for the EXACT package name is the only reliable test.
# Issue #205.
msg_info "Installing $pkg..."
# dpkg-query for the EXACT package name — `dpkg -l | grep -qw python3`
# falsely matches `python3-pip`. Issue #205.
if ! dpkg-query -W -f='${Status}' "$pkg" 2>/dev/null | grep -q "ok installed"; then
if apt-get install -y "$pkg" > /dev/null 2>&1; then
update_config "$pkg" "installed"
@@ -589,11 +648,12 @@ install_beta() {
else
update_config "$pkg" "already_installed"
fi
msg_ok "$pkg ready."
done
msg_ok "Dependencies installed: jq, dialog, curl, git."
# ── Step 2: Clone develop branch ─────────────────────
# ── Step 3: Clone develop branch ─────────────────────
((current_step++))
show_progress $current_step $total_steps "Cloning ProxMenux develop branch"
@@ -612,7 +672,7 @@ install_beta() {
cd "$TEMP_DIR"
# ── Step 3: Files ─────────────────────────────────────
# ── Step 4: Files ─────────────────────────────────────
((current_step++))
show_progress $current_step $total_steps "Creating directories and copying files"
@@ -623,7 +683,11 @@ install_beta() {
mkdir -p "$BASE_DIR/oci"
cp "./scripts/utils.sh" "$UTILS_FILE"
cp "./menu" "$INSTALL_DIR/$MENU_SCRIPT"
# Atomic install of /usr/local/bin/menu — see install_proxmenux.sh
# for the rationale (prevents partial-file reads during mid-update
# parsing).
cp "./menu" "$INSTALL_DIR/${MENU_SCRIPT}.new"
mv -f "$INSTALL_DIR/${MENU_SCRIPT}.new" "$INSTALL_DIR/$MENU_SCRIPT"
cp "./version.txt" "$LOCAL_VERSION_FILE" 2>/dev/null || true
# Store beta version marker
@@ -636,11 +700,23 @@ install_beta() {
cp "./install_proxmenux.sh" "$BASE_DIR/install_proxmenux.sh" 2>/dev/null || true
cp "./install_proxmenux_beta.sh" "$BASE_DIR/install_proxmenux_beta.sh" 2>/dev/null || true
# Pre-built translation cache. The runtime translate() in utils.sh
# reads $BASE_DIR/lang/<lang>.json — these files ship with the repo
# (one per supported language) so every beta install is multilingual
# without any runtime download or Python dependency. Refresh the
# whole dir on every install so a language that was renamed or
# dropped upstream disappears here too.
if [ -d "./lang" ]; then
rm -rf "$BASE_DIR/lang"
mkdir -p "$BASE_DIR/lang"
cp -r "./lang/"* "$BASE_DIR/lang/" 2>/dev/null || true
fi
# Wipe the scripts tree before copying so any file removed upstream
# (renamed, consolidated, deprecated) disappears from the user install.
# Only $BASE_DIR/scripts/ is cleared; config.json, cache.json,
# components_status.json, version.txt, beta_version.txt, monitor.db,
# smart/, oci/ and the AppImage live outside this path and are preserved.
# Only $BASE_DIR/scripts/ is cleared; config.json, components_status.json,
# version.txt, beta_version.txt, monitor.db, smart/, oci/ and the
# AppImage live outside this path and are preserved.
rm -rf "$BASE_DIR/scripts"
mkdir -p "$BASE_DIR/scripts"
cp -r "./scripts/"* "$BASE_DIR/scripts/"
@@ -663,7 +739,7 @@ install_beta() {
msg_ok "Files installed. Beta version: ${beta_version}."
# ── Step 4: Monitor ───────────────────────────────────
# ── Step 5: Monitor ───────────────────────────────────
((current_step++))
show_progress $current_step $total_steps "Installing ProxMenux Monitor (beta)"
@@ -698,6 +774,14 @@ check_stable_available() {
}
# ── Entry point ────────────────────────────────────────────
# Parse --update before any work so the welcome banner can be skipped
# and the relaunch hand-off can fire at the end. The flag arrives from
# `menu`'s check_updates_beta() → `exec bash $INSTALL_SCRIPT --update`.
UPDATE_MODE=0
if [[ "${1:-}" == "--update" ]]; then
UPDATE_MODE=1
fi
if [ "$(id -u)" -ne 0 ]; then
echo -e "${RD}[ERROR] This script must be run as root.${CL}"
exit 1
@@ -705,9 +789,13 @@ fi
cleanup_corrupted_files
show_proxmenux_logo
show_beta_welcome
msg_title "Installing ProxMenux Beta — branch: develop"
if [[ "$UPDATE_MODE" == "1" ]]; then
msg_title "Updating ProxMenux Beta — branch: develop"
else
show_beta_welcome
msg_title "Installing ProxMenux Beta — branch: develop"
fi
install_beta
# Load utils if available
@@ -723,6 +811,17 @@ if [ -x "$BASE_DIR/scripts/global/cleanup_gpu_hookscripts.sh" ]; then
bash "$BASE_DIR/scripts/global/cleanup_gpu_hookscripts.sh" || true
fi
# UPDATE_MODE used to `exec "$INSTALL_DIR/$MENU_SCRIPT"` here to
# auto-relaunch the freshly-installed menu. That produced visible
# "line: syntax" errors when bash tried to read the just-rewritten
# /usr/local/bin/menu (or a shared utils.sh sourced by it) under its
# feet. Fall through to the standard end-of-run message instead —
# the operator types `menu` when ready and the terminal is stable.
#
# `change_release_channel` in scripts/menus/config_menu.sh is
# unaffected: it invokes the installer without `--update`
# (UPDATE_MODE=0) so it never went through this branch.
msg_title "ProxMenux Beta installed successfully"
if systemctl is-active --quiet proxmenux-monitor.service; then
+2169 -42
View File
File diff suppressed because it is too large Load Diff
-198
View File
@@ -1,198 +0,0 @@
{
"Language changed to": {
"es": "Idioma cambiado a",
"fr": "Langue changée en",
"de": "Sprache geändert zu",
"it": "Lingua cambiata in",
"pt": "Idioma alterado para"
},
"Main Menu": {
"es": "Menú principal",
"fr": "Menu principal",
"de": "Hauptmenü",
"it": "Menu principale",
"pt": "Menu principal"
},
"Select an option:": {
"es": "Seleccione una opción:",
"fr": "Sélectionnez une option :",
"de": "Wählen Sie eine Option aus:",
"it": "Selezionare un'opzione:",
"pt": "Selecione uma opção:"
},
"GPUs and Coral-TPU": {
"es": "GPUs y Coral-TPU",
"fr": "GPUs et Coral-TPU",
"de": "GPUs und Coral-TPU",
"it": "GPUs e Coral-TPU",
"pt": "GPUs e Coral-TPU"
},
"Hard Drives, Disk Images, and Storage": {
"es": "Discos duros, imágenes de disco y almacenamiento",
"fr": "Disques durs, images disque et stockage",
"de": "Festplatten, Festplattenabbilder und Speicherplatz",
"it": "Dischi rigidi, immagini del disco e archiviazione",
"pt": "Discos rígidos, imagens de disco e armazenamento"
},
"Network": {
"es": "Red",
"fr": "Réseau",
"de": "Netzwerk",
"it": "Rete",
"pt": "Rede"
},
"Settings": {
"es": "Configuración",
"fr": "Paramètres",
"de": "Einstellungen",
"it": "Impostazioni",
"pt": "Configurações"
},
"Exit": {
"es": "Salir",
"fr": "Quitter",
"de": "Beenden",
"it": "Esci",
"pt": "Sair"
},
"HW: GPUs and Coral": {
"es": "HW: GPUs y Coral",
"fr": "HW: GPUs et Coral",
"de": "HW: GPUs und Coral",
"it": "HW: GPUs e Coral",
"pt": "HW: GPUs e Coral"
},
"Return to Main Menu": {
"es": "Volver al menú principal",
"fr": "Retour au menu principal",
"de": "Zum Hauptmenü zurückkehren",
"it": "Torna al menu principale",
"pt": "Retornar ao menu principal"
},
"Disk and Storage Menu": {
"es": "Menú de discos y almacenamiento",
"fr": "Menu des disques et stockage",
"de": "Datenträger- und Speichermenü",
"it": "Menu dischi e archiviazione",
"pt": "Menu de discos e armazenamento"
},
"Add Disk Passthrough to a VM": {
"es": "Añadir disco Passthrough a una VM",
"fr": "Ajouter un disque Passthrough à une VM",
"de": "Disk-Passthrough zu einer VM hinzufügen",
"it": "Aggiungi passthrough del disco a una VM",
"pt": "Adicionar passthrough de disco a uma VM"
},
"Network Menu": {
"es": "Menú de red",
"fr": "Menu réseau",
"de": "Netzwerkmenü",
"it": "Menu di rete",
"pt": "Menu de rede"
},
"Repair Network": {
"es": "Reparar red",
"fr": "Réparer le réseau",
"de": "Netzwerk reparieren",
"it": "Riparare la rete",
"pt": "Reparar rede"
},
"Configuration Menu": {
"es": "Menú de configuración",
"fr": "Menu de configuration",
"de": "Konfigurationsmenü",
"it": "Menu di configurazione",
"pt": "Menu de configuração"
},
"Change Language": {
"es": "Cambiar idioma",
"fr": "Changer de langue",
"de": "Sprache ändern",
"it": "Cambia lingua",
"pt": "Alterar idioma"
},
"Show Version Information": {
"es": "Mostrar información de la versión",
"fr": "Afficher les informations de version",
"de": "Versionsinformationen anzeigen",
"it": "Mostra informazioni sulla versione",
"pt": "Mostrar informações da versão"
},
"Uninstall ProxMenu": {
"es": "Desinstalar ProxMenu",
"fr": "Désinstaller ProxMenu",
"de": "ProxMenu deinstallieren",
"it": "Disinstallare ProxMenu",
"pt": "Desinstalar ProxMenu"
},
"Select a new language for the menu:": {
"es": "Seleccione un nuevo idioma para el menú:",
"fr": "Sélectionnez une nouvelle langue pour le menu :",
"de": "Wählen Sie eine neue Sprache für das Menü:",
"it": "Seleziona una nuova lingua per il menu:",
"pt": "Selecione um novo idioma para o menu:"
},
"English (Recommended)": {
"es": "Inglés (recomendado)",
"fr": "Anglais (recommandé)",
"de": "Englisch (empfohlen)",
"it": "Inglese (consigliato)",
"pt": "Inglês (recomendado)"
},
"Spanish": {
"es": "Español",
"fr": "Espagnol",
"de": "Spanisch",
"it": "Spagnolo",
"pt": "Espanhol"
},
"French": {
"es": "Francés",
"fr": "Français",
"de": "Französisch",
"it": "Francese",
"pt": "Francês"
},
"German": {
"es": "Alemán",
"fr": "Allemand",
"de": "Deutsch",
"it": "Tedesco",
"pt": "Alemão"
},
"Italian": {
"es": "Italiano",
"fr": "Italien",
"de": "Italienisch",
"it": "Italiano",
"pt": "Italiano"
},
"Portuguese": {
"es": "Portugués",
"fr": "Portugais",
"de": "Portugiesisch",
"it": "Portoghese",
"pt": "Português"
},
"Simplified Chinese": {
"es": "Chino simplificado",
"fr": "Chinois simplifié",
"de": "Vereinfachtes Chinesisch",
"it": "Cinese semplificato",
"pt": "Chinês simplificado"
},
"Japanese": {
"es": "Japonés",
"fr": "Japonais",
"de": "Japanisch",
"it": "Giapponese",
"pt": "Japonês"
},
"Thank you for using ProxMenu. Goodbye!": {
"es": "Gracias por usar ProxMenu. ¡Hasta luego!",
"fr": "Merci d'avoir utilisé ProxMenu. À bientôt !",
"de": "Vielen Dank, dass Sie ProxMenu verwendet haben. Bis bald!",
"it": "Grazie per aver usato ProxMenu. A presto!",
"pt": "Obrigado por usar o ProxMenu. Até logo!"
}
}
+5267
View File
File diff suppressed because it is too large Load Diff
-113
View File
@@ -1,113 +0,0 @@
# General system messages
MAIN_MENU_TITLE="ProxMenux - Main Menu"
CONFIG_TITLE="ProxMenux - Configuration"
SELECT_OPTION="Select an option:"
LANG_OPTION="Change language"
UNINSTALL_OPTION="Uninstall ProxMenu"
EXIT_MENU="Exit"
EXIT_MESSAGE="Exiting menu. Goodbye!"
# Main menu options
OPTION_1="Configure iGPU + TPU"
OPTION_2="Repair network"
OPTION_3="Settings"
# Version messages
VERSION_OPTION="Show version information"
VERSION_TITLE="Version Information"
VERSION_INFO="Current version: %s\n\nFor more information, visit:\nhttps://github.com/MacRimi/ProxMenux"
# Update messages
UPDATE_CHECKING="Checking for updates..."
UPDATE_ERROR_REMOTE="Error checking remote version."
UPDATE_NEW_AVAILABLE="New version available: %s (current: %s)"
UPDATE_TITLE="Update available"
UPDATE_PROMPT="Do you want to update to the latest version %s?"
UPDATE_POSTPONED="Update postponed."
UPDATE_CURRENT="The menu is already up to date (%s)."
UPDATE_PROCESS="Updating to version %s..."
UPDATE_COMPLETE="Update completed to version %s."
UPDATE_ERROR_DOWNLOAD="Error downloading the update."
# Uninstall messages
UNINSTALL_TITLE="Uninstall ProxMenu"
UNINSTALL_CONFIRM="Are you sure you want to uninstall ProxMenu?"
UNINSTALL_COMPLETE="ProxMenu has been uninstalled successfully."
UNINSTALL_PROCESS="Uninstalling ProxMenu..."
# Script messages
SCRIPT_RUNNING="Running igpu_tpu.sh script..."
SCRIPT_SUCCESS="Script executed successfully."
SCRIPT_ERROR="Error executing the script."
# Language messages
LANG_SELECT="Select Language"
LANG_PROMPT="Choose your language:"
LANG_ERROR="No language selected. Exiting..."
LANG_SUCCESS="Selected language:"
LANG_LOADED="Language loaded:"
LANG_DOWNLOAD="Downloading language file..."
LANG_DOWNLOAD_ERROR="Error downloading language file. Check your internet connection."
LANG_EXISTS="Language file exists locally."
# Dependency messages
DEPS_INSTALLING="Installing necessary dependencies..."
DEPS_SUCCESS="Dependencies installed."
DEPS_ERROR="Error installing dependencies. Please install whiptail manually."
# Bilingual messages (first run)
INITIAL_LANG_SELECT="Select Language / Seleccionar Idioma"
INITIAL_LANG_PROMPT="Choose your language / Elige tu idioma:"
INITIAL_LANG_ERROR="No language selected. Exiting... / No se seleccionó ningún idioma. Saliendo..."
# --- Messages for the network repair script (repair_network.sh) ---
REPAIR_MENU_TITLE="Network Repair Menu"
MENU_PROMPT="Please select an option:"
MENU_REPAIR="Repair network"
MENU_VERIFY="Verify network configuration"
MENU_SHOW_IP="Show IP information"
MENU_EXIT="Exit"
MENU_CANCELED="Operation canceled by user."
MENU_EXIT_MSG="Exiting network repair script. Goodbye!"
INVALID_OPTION="Invalid option. Please try again."
PRESS_ENTER="Press Enter to continue..."
RESULT_TITLE="Operation Result"
REPAIR_COMPLETED="Network repair completed."
VERIFY_COMPLETED="Network verification completed."
IP_INFO_COMPLETED="IP information displayed."
NETWORK_ERROR="ERROR"
NETWORK_SUCCESS="SUCCESS"
NETWORK_WARNING="WARNING"
NETWORK_PHYSICAL_INTERFACES="Detected physical interfaces"
NETWORK_CONFIGURED_INTERFACES="Configured interfaces"
NETWORK_CHECKING_BRIDGES="Checking bridge configuration"
NETWORK_BRIDGE_PORT_MISSING="Bridge port non-existent or not active"
NETWORK_BRIDGE_PORT_UPDATED="Bridge port updated"
NETWORK_NO_PHYSICAL_INTERFACE="No suitable physical interface found"
NETWORK_BRIDGE_PORT_OK="Bridge port correct"
NETWORK_CLEANING_INTERFACES="Cleaning configurations of non-existent interfaces"
NETWORK_INTERFACE_REMOVED="Interface removed"
NETWORK_CONFIGURING_INTERFACES="Configuring interfaces"
NETWORK_INTERFACE_ADDED="Interface added"
NETWORK_RESTARTING="Restarting network service"
NETWORK_RESTART_SUCCESS="Network service restarted successfully"
NETWORK_RESTART_FAILED="Failed to restart network service"
NETWORK_CONNECTIVITY_OK="Network connectivity OK"
NETWORK_CONNECTIVITY_FAILED="Network connectivity failed"
NETWORK_IP_INFO="IP Information"
NETWORK_NO_IP="No IP"
NETWORK_REPAIR_STARTED="Starting network repair"
NETWORK_REPAIR_COMPLETED="Network repair completed"
NETWORK_REPAIR_FAILED="Network repair failed"
NETWORK_REPAIR_PROCESS_FINISHED="Network repair process finished"
NETWORK_VERIFY_STARTED="Starting network verification"
NETWORK_VERIFY_FINISHED="Network verification finished"
NETWORK_REPAIR_RUNNING="Running network repair..."
NETWORK_REPAIR_SUCCESS="Network repair executed successfully."
NETWORK_REPAIR_ERROR="Error executing network repair."
NETWORK_VERIFY_RUNNING="Running network verification..."
NETWORK_VERIFY_SUCCESS="Network verification completed successfully."
NETWORK_VERIFY_ERROR="Error executing network verification."
NETWORK_IP_INFO_RUNNING="Obtaining IP information..."
NETWORK_IP_INFO_SUCCESS="IP information obtained successfully."
NETWORK_IP_INFO_ERROR="Error obtaining IP information."
+5267
View File
File diff suppressed because it is too large Load Diff
-115
View File
@@ -1,115 +0,0 @@
# Mensajes generales del sistema
MAIN_MENU_TITLE="ProxMenux - Menú Principal"
CONFIG_TITLE="ProxMenux - Configuración"
SELECT_OPTION="Selecciona una opción:"
LANG_OPTION="Cambiar idioma"
UNINSTALL_OPTION="Desinstalar ProxMenu"
EXIT_MENU="Salir"
EXIT_MESSAGE="Saliendo del menú. ¡Hasta luego!"
# Opciones del menú principal
OPTION_1="Configurar iGPU + TPU"
OPTION_2="Reparar red"
OPTION_3="Configuración"
# Mensajes de versión
VERSION_OPTION="Mostrar información de versión"
VERSION_TITLE="Información de versión"
VERSION_INFO="Versión actual: %s\n\nPara más información, visita:\nhttps://github.com/MacRimi/ProxMenux"
# Mensajes de actualización
UPDATE_CHECKING="Comprobando actualizaciones..."
UPDATE_ERROR_REMOTE="Error al comprobar la versión remota."
UPDATE_NEW_AVAILABLE="Nueva versión disponible: %s (actual: %s)"
UPDATE_TITLE="Actualización disponible"
UPDATE_PROMPT="¿Deseas actualizar a la última versión %s?"
UPDATE_POSTPONED="Actualización pospuesta."
UPDATE_CURRENT="El menú ya está actualizado (%s)."
UPDATE_PROCESS="Actualizando a la versión %s..."
UPDATE_COMPLETE="Actualización completada a la versión %s."
UPDATE_ERROR_DOWNLOAD="Error al descargar la actualización."
# Mensajes de desinstalación
UNINSTALL_TITLE="Desinstalar ProxMenu"
UNINSTALL_CONFIRM="¿Estás seguro de que quieres desinstalar ProxMenu?"
UNINSTALL_COMPLETE="ProxMenu ha sido desinstalado correctamente."
UNINSTALL_PROCESS="Desinstalando ProxMenu..."
# Mensajes de script
SCRIPT_RUNNING="Ejecutando script igpu_tpu.sh..."
SCRIPT_SUCCESS="Script ejecutado correctamente."
SCRIPT_ERROR="Error al ejecutar el script."
# Mensajes de idioma
LANG_SELECT="Seleccionar Idioma"
LANG_PROMPT="Elige tu idioma:"
LANG_ERROR="No se seleccionó ningún idioma. Saliendo..."
LANG_SUCCESS="Idioma seleccionado:"
LANG_LOADED="Idioma cargado:"
LANG_DOWNLOAD="Descargando archivo de idioma..."
LANG_DOWNLOAD_ERROR="Error al descargar el archivo de idioma. Verifica tu conexión a internet."
LANG_EXISTS="Archivo de idioma existe localmente."
# Mensajes de dependencias
DEPS_INSTALLING="Instalando dependencias necesarias..."
DEPS_SUCCESS="Dependencias instaladas."
DEPS_ERROR="Error al instalar dependencias. Por favor, instala whiptail manualmente."
# Mensajes bilingües (primera ejecución)
INITIAL_LANG_SELECT="Seleccionar Idioma / Select Language"
INITIAL_LANG_PROMPT="Elige tu idioma / Choose your language:"
INITIAL_LANG_ERROR="No se seleccionó ningún idioma. Saliendo... / No language selected. Exiting..."
# --- Mensajes para el script de reparación de red (repair_network.sh) ---
REPAIR_MENU_TITLE="Menú de Reparación de Red"
MENU_PROMPT="Por favor, seleccione una opción:"
MENU_REPAIR="Reparar la red"
MENU_VERIFY="Verificar la configuración de red"
MENU_SHOW_IP="Mostrar información de IP"
MENU_EXIT="Salir"
MENU_CANCELED="Operación cancelada por el usuario."
MENU_EXIT_MSG="Saliendo del script de reparación de red. ¡Hasta luego!"
INVALID_OPTION="Opción no válida. Por favor, intente de nuevo."
PRESS_ENTER="Presione Enter para continuar..."
RESULT_TITLE="Resultado de la Operación"
REPAIR_COMPLETED="Reparación de red completada."
VERIFY_COMPLETED="Verificación de red completada."
IP_INFO_COMPLETED="Información de IP mostrada."
NETWORK_ERROR="ERROR"
NETWORK_SUCCESS="ÉXITO"
NETWORK_WARNING="ADVERTENCIA"
NETWORK_PHYSICAL_INTERFACES="Interfaces físicas detectadas"
NETWORK_CONFIGURED_INTERFACES="Interfaces configuradas"
NETWORK_CHECKING_BRIDGES="Verificando configuración de puentes"
NETWORK_BRIDGE_PORT_MISSING="Puerto de puente no existente o no activo"
NETWORK_BRIDGE_PORT_UPDATED="Puerto de puente actualizado"
NETWORK_NO_PHYSICAL_INTERFACE="No se encontró una interfaz física adecuada"
NETWORK_BRIDGE_PORT_OK="Puerto de puente correcto"
NETWORK_CLEANING_INTERFACES="Limpiando configuraciones de interfaces no existentes"
NETWORK_INTERFACE_REMOVED="Interfaz eliminada"
NETWORK_CONFIGURING_INTERFACES="Configurando interfaces"
NETWORK_INTERFACE_ADDED="Interfaz añadida"
NETWORK_RESTARTING="Reiniciando el servicio de red"
NETWORK_RESTART_SUCCESS="Servicio de red reiniciado con éxito"
NETWORK_RESTART_FAILED="Error al reiniciar el servicio de red"
NETWORK_CONNECTIVITY_OK="Conectividad de red OK"
NETWORK_CONNECTIVITY_FAILED="Fallo en la conectividad de red"
NETWORK_IP_INFO="Información de IP"
NETWORK_NO_IP="No IP"
NETWORK_REPAIR_STARTED="Iniciando reparación de red"
NETWORK_REPAIR_COMPLETED="Reparación de red completada"
NETWORK_REPAIR_FAILED="Fallo en la reparación de red"
NETWORK_REPAIR_PROCESS_FINISHED="Proceso de reparación de red finalizado"
NETWORK_VERIFY_STARTED="Iniciando verificación de red"
NETWORK_VERIFY_FINISHED="Verificación de red finalizada"
NETWORK_REPAIR_RUNNING="Ejecutando reparación de red..."
NETWORK_REPAIR_SUCCESS="Reparación de red ejecutada con éxito."
NETWORK_REPAIR_ERROR="Error al ejecutar la reparación de red."
NETWORK_VERIFY_RUNNING="Ejecutando verificación de red..."
NETWORK_VERIFY_SUCCESS="Verificación de red completada con éxito."
NETWORK_VERIFY_ERROR="Error al ejecutar la verificación de red."
NETWORK_IP_INFO_RUNNING="Obteniendo información de IP..."
NETWORK_IP_INFO_SUCCESS="Información de IP obtenida con éxito."
NETWORK_IP_INFO_ERROR="Error al obtener información de IP."
# --- Fin de mensajes para repair_network.sh ---
-1240
View File
File diff suppressed because it is too large Load Diff
+5267
View File
File diff suppressed because it is too large Load Diff
+5267
View File
File diff suppressed because it is too large Load Diff
+5267
View File
File diff suppressed because it is too large Load Diff
+76 -2
View File
@@ -135,8 +135,40 @@ check_updates_stable() {
[[ -z "$LOCAL_VERSION" ]] && return 0
[[ "$LOCAL_VERSION" = "$REMOTE_VERSION" ]] && return 0
if whiptail --title "$(translate 'Update Available')" \
--yesno "$(translate 'New version available') ($REMOTE_VERSION)\n\n$(translate 'Do you want to update now?')" \
# Extract the translated prompt strings into variables FIRST so the
# whiptail line below is trivially parseable. A user on the 1.2.2
# update path hit:
# menu: line 138: syntax error near unexpected token `$REMOTE_VERSION'
# The original inline form was technically valid bash, but a
# translate() return that contains a stray quote or paren is enough
# to confuse a partially-rewritten file (race during update) or a
# corrupted download. Splitting the strings out closes the entire
# parsing-risk surface for zero behavioural change.
local PROMPT_TITLE PROMPT_AVAIL PROMPT_ASK
PROMPT_TITLE="$(translate 'Update Available')"
PROMPT_AVAIL="$(translate 'New version available')"
PROMPT_ASK="$(translate 'Do you want to update now?')"
# Running inside the Monitor's WebSocket terminal: the installer will
# restart the Monitor service, which kills this shell mid-install and
# leaves the update broken. Inform and route to SSH / host console.
if [[ "${PROXMENUX_TERMINAL:-}" == "monitor" ]]; then
local WS_INFO
WS_INFO="$(translate 'A new ProxMenux version is available:') ${REMOTE_VERSION}
$(translate 'This session is running in the Monitor terminal. Updating from here would restart the Monitor service and cut the connection mid-install, leaving the update in a broken state.')
$(translate 'Run the update from an SSH session or the Proxmox host console with:')
bash -c \"\$(wget -qLO - ${INSTALL_URL})\"
$(translate 'You can keep using ProxMenux from this terminal.')"
whiptail --title "$PROMPT_TITLE" --msgbox "$WS_INFO" 20 78
return 0
fi
if whiptail --title "$PROMPT_TITLE" \
--yesno "$PROMPT_AVAIL ($REMOTE_VERSION)\n\n$PROMPT_ASK" \
10 60 --defaultno; then
msg_warn "$(translate 'Starting ProxMenux update...')"
@@ -166,6 +198,20 @@ check_updates_beta() {
[[ -z "$REMOTE_BETA" || -z "$LOCAL_BETA" || "$LOCAL_BETA" = "$REMOTE_BETA" ]] && return 0
[[ "$(printf '%s\n%s\n' "$LOCAL_BETA" "$REMOTE_BETA" | sort -V | tail -1)" = "$REMOTE_BETA" ]] || return 0
if [[ "${PROXMENUX_TERMINAL:-}" == "monitor" ]]; then
whiptail --title "Beta Update Available" --msgbox "\
A new beta build is available: $REMOTE_BETA
This session is running in the Monitor terminal. Updating from here would restart the Monitor service and cut the connection mid-install, leaving the update in a broken state.
Run the update from an SSH session or the Proxmox host console with:
bash -c \"\$(wget -qLO - $REPO_DEVELOP/install_proxmenux_beta.sh)\"
You can keep using ProxMenux from this terminal." 20 78
return 0
fi
if whiptail --title "Beta Update Available" \
--yesno "A new beta build is available!\n\nInstalled beta : $LOCAL_BETA\nNew beta build : $REMOTE_BETA\n\nDo you want to update now?" \
12 64 --defaultno; then
@@ -189,6 +235,34 @@ main_menu() {
exec bash "$MAIN_MENU"
}
# `menu -v` / `-h` print info and exit without opening the TUI so
# admins can query the install over SSH (issue #240).
case "${1:-}" in
-v|--version|-V)
local_ver="unknown"
[[ -f "$LOCAL_VERSION_FILE" ]] && local_ver="$(<"$LOCAL_VERSION_FILE")"
printf 'ProxMenux %s\n' "$local_ver"
if is_beta && [[ -f "$BETA_VERSION_FILE" ]]; then
printf 'Beta build: %s\n' "$(<"$BETA_VERSION_FILE")"
fi
exit 0
;;
-h|--help)
cat <<'EOF'
Usage: menu [OPTION]
Interactive menu for Proxmox VE management.
Options:
-v, --version Print installed ProxMenux version and exit.
-h, --help Show this help and exit.
Run without arguments to launch the interactive menu.
EOF
exit 0
;;
esac
load_language
initialize_cache
auto_repair_monitor_unit
+714
View File
@@ -0,0 +1,714 @@
#!/bin/bash
# ==========================================================
# ProxMenux - Apply Cluster Configs (post-boot)
# ==========================================================
# Fires AFTER pve-cluster.service is up, when /etc/pve is
# the live pmxcfs FUSE mount. We can write individual files
# to /etc/pve at this point and they propagate through the
# cluster filesystem normally — no need to stop pve-cluster
# (which would be unsafe at this stage of boot).
#
# Trigger: apply_pending_restore.sh writes a marker file at
# /var/lib/proxmenux/cluster-apply-pending whose contents is
# the absolute path of the recovery dir containing the
# extracted /etc/pve content. The systemd unit has
# ConditionPathExists=<marker>, so on a normal boot (no
# marker), the unit short-circuits and does nothing.
set +u
MARKER="${PMX_CLUSTER_APPLY_MARKER:-/var/lib/proxmenux/cluster-apply-pending}"
LOG_DIR="${PMX_LOG_DIR:-/var/log/proxmenux}"
# State file the Monitor Web polls to show a live progress card on
# the Backups tab. A dismiss action (POST /api/host-backups/restore/dismiss)
# just flips `acknowledged` to true — the file itself lives until the
# next restore overwrites it, and a copy is archived under history/
# when the run finishes so the operator can browse past restores.
STATE_DIR="/var/lib/proxmenux"
STATE_FILE="$STATE_DIR/restore-state.json"
HISTORY_DIR="$STATE_DIR/restore-history"
mkdir -p "$STATE_DIR" "$HISTORY_DIR" >/dev/null 2>&1 || true
mkdir -p "$LOG_DIR" >/dev/null 2>&1 || true
LOG_FILE="${LOG_DIR}/proxmenux-cluster-postboot-$(date +%Y%m%d_%H%M%S).log"
exec >>"$LOG_FILE" 2>&1
# Capture start epoch BEFORE any long-running step. The final duration
# is derived from this; the previous approach used stat -c %Y on the
# log file, which reads the last-write mtime — and since we `exec >>`
# to the log for the whole run, that mtime is always ~end-of-run and
# the duration came out as 0m00s.
POSTBOOT_START_EPOCH=$(date +%s)
# ── State-file helpers ─────────────────────────────────────────
# Every milestone advances `steps_done` and optionally updates a
# handful of other fields. Writes go through a temp file + rename
# so the Monitor never reads a half-written JSON. All calls are
# `|| true` at the callsite — if jq or write fails, the restore
# still proceeds; only the UI progress reporting suffers.
_state_started_at="$(date -Iseconds)"
_state_steps_total=0
_state_steps_done=0
_state_write() {
# Merges a JSON snippet ($1) into the existing state file.
# Missing state file → seeded from an empty JSON object first.
command -v jq >/dev/null 2>&1 || return 0
[[ -f "$STATE_FILE" ]] || echo '{}' > "$STATE_FILE"
local tmp
tmp=$(mktemp "${STATE_FILE}.XXXXXX") || return 0
if jq -c ". * $1" "$STATE_FILE" > "$tmp" 2>/dev/null; then
mv -f "$tmp" "$STATE_FILE"
else
rm -f "$tmp"
fi
}
_state_step() {
# Called as a *step transition*: the previous step just finished,
# start the next one. Increments steps_done and sets the label to
# $1. Init seeds current_step with the first step's label; every
# _state_step call after that advances to the NEXT step.
#
# Example flow with 3 total steps:
# init → steps_done=0, current_step="Applying cluster config"
# step "Foo" → steps_done=1, current_step="Foo"
# step "Bar" → steps_done=2, current_step="Bar"
# _state_finish→ steps_done=3, status="complete"
local label="$1"
_state_steps_done=$((_state_steps_done + 1))
# jq variable names sdone/stotal — plain `done` collides with the
# bash reserved word when the arg is on its own continuation line.
_state_write "$(jq -n \
--arg step "$label" \
--argjson sdone "$_state_steps_done" \
--argjson stotal "$_state_steps_total" \
'{current_step:$step, steps_done:$sdone, steps_total:$stotal}')"
}
_state_component() {
# Add or update an entry in state.components. Arrays are replaced
# by jq's `*` merge, so plain _state_write can't be used for the
# per-component list — this helper explicitly appends new entries
# and updates existing ones by name.
# $1 = component name (nvidia_driver, coral_driver, …)
# $2 = status: installing | ok | failed
# $3 = per-component log path (empty allowed)
# $4 = installer exit code (only meaningful for failed; empty otherwise)
command -v jq >/dev/null 2>&1 || return 0
[[ -f "$STATE_FILE" ]] || echo '{}' > "$STATE_FILE"
local tmp
tmp=$(mktemp "${STATE_FILE}.XXXXXX") || return 0
if jq -c \
--arg name "$1" \
--arg status "$2" \
--arg log "$3" \
--arg rc "${4:-}" \
'($.components // []) as $comps
| ($comps | map(.name) | index($name)) as $idx
| ({name:$name, status:$status, log:$log}
+ (if $rc == "" then {} else {exit_code:$rc} end)) as $entry
| .components =
(if $idx == null then $comps + [$entry]
else ($comps | map(if .name == $name then $entry else . end)) end)' \
"$STATE_FILE" > "$tmp" 2>/dev/null; then
mv -f "$tmp" "$STATE_FILE"
else
rm -f "$tmp"
fi
}
_state_finish() {
# $1 = "complete" | "failed"
# Promotes steps_done to steps_total so the progress bar reads 100%
# instead of freezing at whatever the last _state_step call left it
# at (typically N-1 because "finalize" itself never gets its own
# _state_step). Also relabels current_step so the card stops
# showing the last in-flight step ("Boot sanity check") as if it
# were still running.
local final_status="$1"
local final_label="Restore finished"
[[ "$final_status" == "failed" ]] && final_label="Restore failed"
_state_write "$(jq -n \
--arg s "$final_status" \
--arg t "$(date -Iseconds)" \
--arg dur "${POSTBOOT_DURATION_FMT:-}" \
--arg step "$final_label" \
--argjson stotal "$_state_steps_total" \
'{status:$s, finished_at:$t, duration:$dur, current_step:$step, steps_done:$stotal, steps_total:$stotal}')"
# Archive a copy to history/ so past restores stay browsable
# from the Monitor even after the operator dismisses the card.
if [[ -f "$STATE_FILE" ]]; then
cp -f "$STATE_FILE" "$HISTORY_DIR/$(date +%Y%m%d_%H%M%S)-${final_status}.json" 2>/dev/null || true
# Keep the last 20 entries in the history dir. Everything
# older is expected to have been reviewed already. Using
# find+sort-by-mtime keeps this safe against odd filenames.
find "$HISTORY_DIR" -maxdepth 1 -type f -name '*.json' -printf '%T@ %p\n' 2>/dev/null \
| sort -rn | tail -n +21 | cut -d' ' -f2- | xargs -r rm -f 2>/dev/null || true
fi
}
echo "=== ProxMenux cluster post-boot apply at $(date -Iseconds) ==="
if [[ ! -f "$MARKER" ]]; then
echo "No marker found at $MARKER — nothing to apply."
exit 0
fi
# Marker is env-style key=value, written by apply_pending_restore.sh.
# Defaults so a malformed marker still gives us safe behaviour.
RECOVERY_ROOT=""
PENDING_DIR=""
NEEDS_INITRAMFS=0
NEEDS_GRUB=0
# shellcheck source=/dev/null
source "$MARKER"
echo "Recovery root: $RECOVERY_ROOT"
echo "Pending dir: $PENDING_DIR"
echo "Needs initramfs: $NEEDS_INITRAMFS"
echo "Needs grub: $NEEDS_GRUB"
# Compute how many milestones the Monitor will see for this run.
# Base 3 = apply /etc/pve + boot sanity check + finalize. Add one
# for each optional phase that will actually run.
_state_steps_total=3
[[ "$NEEDS_INITRAMFS" == "1" ]] && _state_steps_total=$((_state_steps_total + 1))
[[ "$NEEDS_GRUB" == "1" ]] && _state_steps_total=$((_state_steps_total + 1))
# Count components that will be re-installed via --auto-reinstall.
# Read the same components_status.json the reinstall loop uses so
# our step count matches exactly what the loop will process.
_COMP_STATUS_PATH="/usr/local/share/proxmenux/components_status.json"
_COMPONENT_KEYS=(nvidia_driver amdgpu_top intel_gpu_tools coral_driver)
_comps_to_reinstall=0
if command -v jq >/dev/null 2>&1 && [[ -f "$_COMP_STATUS_PATH" ]]; then
for _k in "${_COMPONENT_KEYS[@]}"; do
[[ "$(jq -r ".$_k.status // \"\"" "$_COMP_STATUS_PATH" 2>/dev/null)" == "installed" ]] \
&& _comps_to_reinstall=$((_comps_to_reinstall + 1))
done
fi
(( _comps_to_reinstall > 0 )) && _state_steps_total=$((_state_steps_total + _comps_to_reinstall))
# Seed the initial state file so the Monitor's poll sees the run
# almost immediately after the postboot unit starts. `acknowledged`
# false means the Backups tab card will show; the operator flips
# it to true (via POST /dismiss) after they've read the summary.
_state_write "$(jq -n \
--arg started "$_state_started_at" \
--arg log "$LOG_FILE" \
--argjson steps "$_state_steps_total" \
'{status:"running",
started_at:$started,
finished_at:null,
current_step:"Applying cluster config",
steps_done:0,
steps_total:$steps,
log_path:$log,
components:[],
rollback_delta:{},
sanity_warnings:[],
summary:null,
acknowledged:false}')"
if [[ -z "$RECOVERY_ROOT" || ! -d "$RECOVERY_ROOT" ]]; then
echo "Recovery root invalid — aborting cleanly."
rm -f "$MARKER"
_state_finish "failed" || true
exit 0
fi
SOURCE_PVE="$RECOVERY_ROOT/etc/pve"
if [[ ! -d "$SOURCE_PVE" ]]; then
echo "No /etc/pve content in recovery dir — nothing to do."
rm -f "$MARKER"
exit 0
fi
# Wait for pmxcfs to be fully writable. The After=pve-cluster.service
# in our unit gets us past the service-start point, but on slow boots
# the FUSE mount can take a few extra seconds to settle.
echo "Waiting for /etc/pve to be writable..."
for i in {1..60}; do
if [[ -d /etc/pve ]] \
&& touch "/etc/pve/.proxmenux-test-$$" 2>/dev/null; then
rm -f "/etc/pve/.proxmenux-test-$$" 2>/dev/null
echo "/etc/pve writable after ${i}s"
break
fi
sleep 1
done
# ── Detect source node name for cross-host node rename ────
# The source backup's node dir is whatever the source host
# was called; we copy its contents into THIS host's node
# dir. Two sources for the source hostname, in order of
# preference:
# 1. metadata/run_info.env from the pending dir (definitive)
# 2. The first (and usually only) dir under nodes/ in the
# source backup — works when metadata is missing
SRC_NODE=""
if [[ -n "$PENDING_DIR" ]]; then
META_RUN_INFO=$(find "$PENDING_DIR" -maxdepth 3 -name run_info.env 2>/dev/null | head -1)
if [[ -n "$META_RUN_INFO" && -f "$META_RUN_INFO" ]]; then
SRC_NODE=$(grep -m1 '^hostname=' "$META_RUN_INFO" 2>/dev/null | cut -d= -f2- | tr -d '[:space:]')
fi
fi
if [[ -z "$SRC_NODE" && -d "$SOURCE_PVE/nodes" ]]; then
SRC_NODE=$(find "$SOURCE_PVE/nodes" -mindepth 1 -maxdepth 1 -type d 2>/dev/null | head -1)
SRC_NODE=$(basename "$SRC_NODE" 2>/dev/null)
fi
CUR_NODE=$(hostname)
echo "Source node: ${SRC_NODE:-(unknown)} / Current node: ${CUR_NODE}"
# ── Apply EVERY top-level file in /etc/pve ────────────────
# Anything that's a regular file at the root of /etc/pve
# (datacenter.cfg, storage.cfg, user.cfg, domains.cfg,
# vzdump.cron, jobs.cfg, replication.cfg, ceph.conf,
# corosync.conf if cluster, etc). pmxcfs symlinks like
# /etc/pve/local, /etc/pve/lxc, /etc/pve/qemu-server,
# /etc/pve/openvz are auto-created by pmxcfs and we skip
# them — copying over them throws "Operation not permitted".
echo ""
echo "── Global config files ──"
copied_global=0
PMX_SYMLINKS_SKIP="local lxc qemu-server openvz"
for src in "$SOURCE_PVE"/*; do
[[ -f "$src" ]] || continue
name=$(basename "$src")
# Skip files that mirror pmxcfs symlinks
skip=0
for s in $PMX_SYMLINKS_SKIP; do
[[ "$name" == "$s" ]] && { skip=1; break; }
done
(( skip )) && continue
if cp -f "$src" "/etc/pve/$name" 2>&1; then
echo "$name"
((copied_global++))
else
echo "$name (cp failed)"
fi
done
# ── Subdirectories we want to preserve verbatim ───────────
# Each gets contents copied flat (no recursive dir copy of
# symlinks). These are the "shared cluster state" dirs.
echo ""
echo "── Cluster subdirectories ──"
copied_subdirs=0
for subdir in firewall sdn mapping virtual-guest priv ha; do
src_dir="$SOURCE_PVE/$subdir"
[[ -d "$src_dir" ]] || continue
mkdir -p "/etc/pve/$subdir" 2>/dev/null || true
while IFS= read -r f; do
rel="${f#"$src_dir"/}"
dst="/etc/pve/$subdir/$rel"
if [[ -d "$f" ]]; then
mkdir -p "$dst" 2>/dev/null || true
elif [[ -f "$f" ]]; then
mkdir -p "$(dirname "$dst")" 2>/dev/null || true
cp -f "$f" "$dst" 2>/dev/null && ((copied_subdirs++))
fi
done < <(find "$src_dir" -mindepth 1 2>/dev/null)
echo "$subdir/ (subtree)"
done
# ── Apply guest configs into THIS node's dir ──────────────
# This is the bit that makes `pct list` / `qm list` show
# the restored guests. We deliberately copy from the
# source's node dir into the current host's node dir, so
# cross-host restores Just Work without renaming anything.
echo ""
echo "── Guest configs (LXC + QEMU) ──"
copied_guests=0
skipped_guests=0
if [[ -n "$SRC_NODE" ]] && [[ -d "$SOURCE_PVE/nodes/$SRC_NODE" ]]; then
for kind in lxc qemu-server; do
src_dir="$SOURCE_PVE/nodes/$SRC_NODE/$kind"
dst_dir="/etc/pve/nodes/$CUR_NODE/$kind"
[[ -d "$src_dir" ]] || continue
mkdir -p "$dst_dir" 2>/dev/null || true
for conf in "$src_dir"/*.conf; do
[[ -f "$conf" ]] || continue
vmid=$(basename "$conf" .conf)
if [[ -e "$dst_dir/$vmid.conf" ]]; then
echo "$kind/$vmid.conf already exists on this host — skipping (avoid clash)"
((skipped_guests++))
continue
fi
if cp -f "$conf" "$dst_dir/$vmid.conf" 2>&1; then
echo "$kind/$vmid.conf"
((copied_guests++))
else
echo "$kind/$vmid.conf (cp failed)"
fi
done
done
else
echo " (no source node dir to copy from)"
fi
# ── LXC bind-mount stub directories ───────────────────────
# LXC containers with `mp<n>: /path,mp=...` bind-mount entries fail the
# pre-start hook (status 2) if `/path` doesn't exist on the host. After a
# cross-host restore the source's bind-mount paths (custom NAS mounts, second
# disk paths, etc.) generally don't exist on the target's fresh install yet.
# We create empty stubs so `onboot: 1` containers start; the operator wires
# the real data source afterwards. PVE-managed storages (`/mnt/pve/*`) and
# /dev/* are skipped — PVE handles the first, kernel handles the second.
echo ""
echo "── LXC bind-mount stubs ──"
stub_created=0
stub_skipped=0
if compgen -G "/etc/pve/nodes/$CUR_NODE/lxc/*.conf" >/dev/null 2>&1; then
for conf in /etc/pve/nodes/"$CUR_NODE"/lxc/*.conf; do
[[ -f "$conf" ]] || continue
while IFS= read -r line; do
if [[ "$line" =~ ^mp[0-9]+:[[:space:]]*(/[^,]+), ]]; then
src="${BASH_REMATCH[1]}"
[[ "$src" == /mnt/pve/* ]] && continue
[[ "$src" == /dev/* ]] && continue
if [[ -e "$src" ]]; then
((stub_skipped++))
continue
fi
if mkdir -p "$src" 2>/dev/null; then
echo " + stub $src (from $(basename "$conf"))"
((stub_created++))
fi
fi
done < "$conf"
done
fi
echo "Stubs: created=$stub_created, already-present=$stub_skipped"
# ── Stale node-dir cleanup ────────────────────────────────
# Fresh PVE install creates /etc/pve/nodes/<install-hostname>/. After our
# restore changes the hostname back to the source's, pve-cluster boots into
# the source's node dir but leaves the install-hostname dir orphaned. The
# web UI then shows a phantom offline node. Only remove dirs whose lxc/
# qemu-server/ are empty — never trample a real second cluster member.
echo ""
echo "── Stale node-dir cleanup ──"
removed_nodes=0
for nodedir in /etc/pve/nodes/*/; do
n=$(basename "$nodedir")
[[ "$n" == "$CUR_NODE" ]] && continue
lxc_empty=1; qemu_empty=1
[[ -d "$nodedir/lxc" ]] && [[ -n "$(ls -A "$nodedir/lxc" 2>/dev/null)" ]] && lxc_empty=0
[[ -d "$nodedir/qemu-server" ]] && [[ -n "$(ls -A "$nodedir/qemu-server" 2>/dev/null)" ]] && qemu_empty=0
if (( lxc_empty && qemu_empty )); then
if rm -rf "$nodedir" 2>/dev/null; then
echo " ✓ removed stale node dir: $n"
((removed_nodes++))
else
echo " ✗ rm failed for $n (pmxcfs may have it busy)"
fi
else
echo " ⚠ kept $n (has guest configs — looks like a real cluster member)"
fi
done
echo "Stale node dirs removed: $removed_nodes"
# ── Done with cluster config apply ─────────────────────────
echo ""
echo "Cluster summary: globals=$copied_global, subdirs=$copied_subdirs, guests=$copied_guests, guest-clashes-skipped=$skipped_guests"
# Remove the marker NOW (before the slow maintenance step
# below) so if the operator reboots mid-maintenance, we
# don't redo the (idempotent but wasteful) cluster apply.
# Maintenance below is also idempotent on re-run but takes
# 10+ min, so we'd rather not repeat it either — see the
# marker handling in the maintenance block.
rm -f "$MARKER"
# ── Post-restore maintenance (slow, deferrable) ────────────
# After a host-config restore, we need to:
# - update-initramfs -u -k all → so /etc/modules /etc/modprobe.d
# /etc/initramfs-tools changes get baked into the initramfs
# of every installed kernel for the NEXT boot.
# - update-grub → so /etc/default/grub changes land in
# /boot/grub/grub.cfg for the NEXT boot.
#
# These are EXPENSIVE (initramfs build per kernel × 3 = 5-10 min;
# grub a few seconds) but the user's system is already fully up
# at this point: they can SSH in, use PVE, do anything — these
# run in the background and finish whenever they finish. The
# unit's TimeoutStartSec=900 (set in apply_pending_restore.sh)
# gives us a 15-min cushion. We log progress to the same log
# file so the operator can `tail -f` if curious.
echo ""
echo "── Post-restore maintenance ──"
# Only do these if the apply_pending_restore.sh's path-trigger
# analysis said they're needed. On a restore that didn't touch
# /etc/modules /etc/default/grub etc., both flags are 0 and we
# skip the slow rebuild entirely.
MAINT_MARKER="/var/lib/proxmenux/post-restore-maintenance-pending"
if [[ "$NEEDS_INITRAMFS" == "1" ]] || [[ "$NEEDS_GRUB" == "1" ]]; then
mkdir -p /var/lib/proxmenux >/dev/null 2>&1 || true
printf 'started: %s\n' "$(date -Iseconds)" > "$MAINT_MARKER"
fi
if [[ "$NEEDS_INITRAMFS" == "1" ]] && command -v update-initramfs >/dev/null 2>&1; then
_state_step "Rebuilding initramfs" || true
echo "Running: update-initramfs -u -k all (5-10 min — restore touched initramfs inputs)"
if update-initramfs -u -k all 2>&1 | tail -10; then
echo " ✓ update-initramfs done"
else
echo " ✗ update-initramfs failed (system still boots; re-run manually)"
fi
if command -v proxmox-boot-tool >/dev/null 2>&1; then
proxmox-boot-tool refresh 2>&1 | tail -3 || true
fi
else
echo "Skipping update-initramfs (restore didn't touch modules/initramfs-tools/crypttab)"
fi
if [[ "$NEEDS_GRUB" == "1" ]] && command -v update-grub >/dev/null 2>&1; then
_state_step "Updating bootloader" || true
echo "Running: update-grub"
if update-grub 2>&1 | tail -3; then
echo " ✓ update-grub done"
else
echo " ✗ update-grub failed (re-run manually)"
fi
else
echo "Skipping update-grub (restore didn't touch /etc/default/grub or /etc/kernel)"
fi
# Clean up the maintenance marker now that we're done.
rm -f "$MAINT_MARKER"
# ── Component auto-reinstall (driven by components_status.json) ──
# The host-config restore brings back ProxMenux state (including
# components_status.json) but NOT the binary artifacts those
# components installed outside of apt — driver modules under
# /lib/modules/<kernel>/, binaries in /usr/bin/<tool>, downloaded
# .deb files, DKMS source trees, etc. For each component the
# restore state says was installed, we kick off its native
# installer in `--auto-reinstall` mode so it replays the install
# without dialogs. The installer's own logic handles "already
# present → no-op", so this is idempotent.
#
# Apt-only components are still handled by the
# packages.manual.list pass done earlier in the restore flow
# (they're in `apt-mark showmanual`). Running the installer here
# for them is harmless overhead (the installer just sees the
# package is present and exits 0), so we don't try to filter.
#
# To register a NEW component for auto-reinstall: add it to the
# COMPONENT_INSTALLERS array below as "component_key:relative
# script path". The script must accept `--auto-reinstall` and
# read its own state from components_status.json.
COMPONENTS_STATUS="/usr/local/share/proxmenux/components_status.json"
COMPONENT_INSTALLERS=(
"nvidia_driver:gpu_tpu/nvidia_installer.sh"
"amdgpu_top:gpu_tpu/amd_gpu_tools.sh"
"intel_gpu_tools:gpu_tpu/intel_gpu_tools.sh"
"coral_driver:gpu_tpu/install_coral.sh"
)
if command -v jq >/dev/null 2>&1 && [[ -f "$COMPONENTS_STATUS" ]]; then
echo ""
echo "── Component auto-reinstall ──"
SCRIPTS_BASE="/usr/local/share/proxmenux/scripts"
for entry in "${COMPONENT_INSTALLERS[@]}"; do
comp="${entry%%:*}"
installer="$SCRIPTS_BASE/${entry#*:}"
comp_status=$(jq -r ".${comp}.status // \"\"" "$COMPONENTS_STATUS" 2>/dev/null)
if [[ "$comp_status" != "installed" ]]; then
continue # Was never installed on the source, or was uninstalled — skip.
fi
if [[ ! -f "$installer" ]]; then
echo "$comp: installer missing at $installer — skipping"
continue
fi
echo ""
echo "$comp (running $installer --auto-reinstall)"
_state_step "Reinstalling $comp" || true
_state_component "$comp" "installing" "" ""
# Redirect to a per-component log instead of piping. NVIDIA's
# runfile installer forks helpers that inherit stdout, so a
# `bash $installer | sed | tail` pipeline never sees EOF after
# the parent exits and hangs until systemd kills the unit.
comp_log="/var/log/proxmenux/component-${comp}-$(date +%Y%m%d_%H%M%S).log"
bash "$installer" --auto-reinstall >"$comp_log" 2>&1
rc=$?
sed -e 's/^/ /' "$comp_log" | tail -15
if (( rc == 0 )); then
echo "$comp ok (full log: $comp_log)"
_state_component "$comp" "ok" "$comp_log" ""
else
echo "$comp installer exited $rc — see $comp_log"
_state_component "$comp" "failed" "$comp_log" "$rc"
fi
done
fi
POSTBOOT_END_EPOCH=$(date +%s)
POSTBOOT_DURATION=$((POSTBOOT_END_EPOCH - POSTBOOT_START_EPOCH))
POSTBOOT_DURATION_FMT=$(printf '%dm%02ds' $((POSTBOOT_DURATION / 60)) $((POSTBOOT_DURATION % 60)))
# ── Rollback delta report (read-only) ──────────────────────────
# If _rs_prepare_pending_restore left a rollback.json in the
# pending dir, surface the deltas that a full rollback would
# touch: VMs/LXCs created after the backup that are still here,
# extra components not present in the backup. The operator
# decides what to do with them — we don't destroy guests or
# uninstall packages automatically (left to a future R2.D fase 2
# once every installer ships an --auto-uninstall mode).
ROLLBACK_PLAN_FILE=""
[[ -n "$PENDING_DIR" && -f "$PENDING_DIR/rollback.json" ]] && \
ROLLBACK_PLAN_FILE="$PENDING_DIR/rollback.json"
if [[ -n "$ROLLBACK_PLAN_FILE" ]] && command -v jq >/dev/null 2>&1; then
rb_vm_extras=$(jq -r '.vms_to_remove | join(", ") // ""' "$ROLLBACK_PLAN_FILE" 2>/dev/null)
rb_lxc_extras=$(jq -r '.lxcs_to_remove | join(", ") // ""' "$ROLLBACK_PLAN_FILE" 2>/dev/null)
rb_comp_extras=$(jq -r '.components_to_uninstall | join(", ") // ""' "$ROLLBACK_PLAN_FILE" 2>/dev/null)
# Copy the structured plan into the state file so the Monitor's
# detail modal can render each list as its own table without
# re-parsing the CSV strings above.
_state_write "$(jq -c '{rollback_delta:{
vms_to_remove: (.vms_to_remove // []),
lxcs_to_remove: (.lxcs_to_remove // []),
components_to_uninstall: (.components_to_uninstall // [])
}}' "$ROLLBACK_PLAN_FILE" 2>/dev/null)" || true
if [[ -n "$rb_vm_extras" || -n "$rb_lxc_extras" || -n "$rb_comp_extras" ]]; then
echo ""
echo "── Rollback delta report ──"
echo "These entries exist on the host but were NOT in the restored backup."
echo "Review them and remove manually if a clean rollback is desired:"
[[ -n "$rb_vm_extras" ]] && echo " • VMs created after the backup: $rb_vm_extras"
[[ -n "$rb_lxc_extras" ]] && echo " • LXCs created after the backup: $rb_lxc_extras"
[[ -n "$rb_comp_extras" ]] && echo " • Components installed after the backup: $rb_comp_extras"
echo ""
echo "Manual cleanup commands (run only if you want to truly roll back):"
for _id in $(echo "$rb_vm_extras" | tr ',' ' '); do
[[ -n "$_id" ]] && echo " qm stop $_id 2>/dev/null; qm destroy $_id --purge"
done
for _id in $(echo "$rb_lxc_extras" | tr ',' ' '); do
[[ -n "$_id" ]] && echo " pct stop $_id 2>/dev/null; pct destroy $_id --purge"
done
fi
fi
# ── Boot sanity check ──────────────────────────────────────────
# Cross-version restores are the most likely path to a broken
# boot: even with the safe-restore filter, catch situations where
# the default kernel has no matching /lib/modules, no ESP is
# configured, or a stale reference survived. Runs on every restore
# (cheap); warnings feed the completion notification so it tells
# the truth instead of a blanket "all fine".
_state_step "Boot sanity check" || true
SANITY_WARNINGS=""
_sanity_warn() {
if [[ -z "$SANITY_WARNINGS" ]]; then
SANITY_WARNINGS="$1"
else
SANITY_WARNINGS="${SANITY_WARNINGS}; $1"
fi
}
if command -v proxmox-boot-tool >/dev/null 2>&1; then
if ! proxmox-boot-tool status 2>/dev/null | grep -q 'configured with'; then
_sanity_warn "proxmox-boot-tool reports no ESP configured"
fi
fi
if [[ -d /boot ]]; then
for _vmlinuz in /boot/vmlinuz-*; do
[[ -e "$_vmlinuz" ]] || continue
_kver="${_vmlinuz##*vmlinuz-}"
if [[ ! -d "/lib/modules/$_kver" ]]; then
_sanity_warn "kernel $_kver has no /lib/modules"
fi
done
fi
if [[ -L /vmlinuz && ! -e /vmlinuz ]]; then
_sanity_warn "/vmlinuz symlink is dangling"
fi
if [[ -n "$SANITY_WARNINGS" ]]; then
echo ""
echo "── Boot sanity check ──"
echo "$SANITY_WARNINGS"
echo "Cross-version: ${HB_COMPAT_CROSS_VERSION:-0}"
fi
# Persist sanity warnings as a JSON array so the Monitor's detail
# modal can render each warning as its own line. Empty warnings →
# empty array, which the UI treats as "sanity OK".
if command -v jq >/dev/null 2>&1; then
_state_write "$(jq -cn --arg s "$SANITY_WARNINGS" \
'{sanity_warnings: (if $s == "" then [] else ($s | split("; ")) end)}')" || true
fi
# ── Notify ProxMenux Monitor that we're done ───────────────────
# Routes through the user's configured channels (Telegram, Discord,
# ntfy, etc.). Localhost-only endpoint, no auth needed. We try
# briefly — if the Monitor isn't running, just log and move on.
COMPONENTS_REINSTALLED_CSV=""
if command -v jq >/dev/null 2>&1 && [[ -f "$COMPONENTS_STATUS" ]]; then
COMPONENTS_REINSTALLED_CSV=$(
for entry in "${COMPONENT_INSTALLERS[@]}"; do
comp="${entry%%:*}"
s=$(jq -r ".${comp}.status // \"\"" "$COMPONENTS_STATUS" 2>/dev/null)
[[ "$s" == "installed" ]] && printf '%s,' "$comp"
done | sed 's/,$//'
)
[[ -z "$COMPONENTS_REINSTALLED_CSV" ]] && COMPONENTS_REINSTALLED_CSV="none"
fi
if command -v curl >/dev/null 2>&1; then
# jq builds a proper JSON when available so SANITY_WARNINGS with
# special chars can't break the payload. Falls back to printf on
# hosts without jq — we already restrict SANITY_WARNINGS content
# to plain ASCII in the sanity check above, so the fallback is
# safe too.
if command -v jq >/dev/null 2>&1; then
PAYLOAD=$(jq -cn \
--arg hostname "$(hostname)" \
--arg guests "${copied_guests:-0}" \
--arg stubs "${stub_created:-0}" \
--arg stale_nodes "${removed_nodes:-0}" \
--arg components "${COMPONENTS_REINSTALLED_CSV:-none}" \
--arg duration "$POSTBOOT_DURATION_FMT" \
--arg warnings "$SANITY_WARNINGS" \
'{hostname:$hostname, guests:$guests, stubs:$stubs, stale_nodes:$stale_nodes, components:$components, duration:$duration, warnings:$warnings}')
else
PAYLOAD=$(printf '{"hostname":"%s","guests":"%s","stubs":"%s","stale_nodes":"%s","components":"%s","duration":"%s","warnings":"%s"}' \
"$(hostname)" \
"${copied_guests:-0}" \
"${stub_created:-0}" \
"${removed_nodes:-0}" \
"${COMPONENTS_REINSTALLED_CSV:-none}" \
"$POSTBOOT_DURATION_FMT" \
"$SANITY_WARNINGS")
fi
NOTIFY_HTTP=$(curl -s -o /dev/null -w '%{http_code}' \
-X POST "http://127.0.0.1:8008/api/internal/restore-event" \
-H "Content-Type: application/json" \
-d "$PAYLOAD" \
--max-time 5 2>/dev/null || echo "000")
if [[ "$NOTIFY_HTTP" == "200" ]]; then
echo "Notification sent (HTTP 200)"
else
echo "Notification skipped (Monitor not reachable or disabled — HTTP $NOTIFY_HTTP)"
fi
fi
# Persist a compact summary the Monitor's card can render inline.
# Mirrors what the notification block sends; keeps a single source
# of truth for both the alert channels and the Web UI.
if command -v jq >/dev/null 2>&1; then
_state_write "$(jq -cn \
--arg hostname "$(hostname)" \
--arg guests "${copied_guests:-0}" \
--arg stubs "${stub_created:-0}" \
--arg stale_nodes "${removed_nodes:-0}" \
--arg components "${COMPONENTS_REINSTALLED_CSV:-none}" \
--arg duration "$POSTBOOT_DURATION_FMT" \
'{summary:{hostname:$hostname, guests:$guests, stubs:$stubs, stale_nodes:$stale_nodes, components:$components, duration:$duration}}')" || true
fi
_state_finish "complete" || true
echo ""
echo "=== Apply finished at $(date -Iseconds) — total ${POSTBOOT_DURATION_FMT} ==="
echo "Log: $LOG_FILE"
+207 -12
View File
@@ -7,8 +7,8 @@ PENDING_BASE="${PMX_RESTORE_PENDING_BASE:-/var/lib/proxmenux/restore-pending}"
CURRENT_LINK="${PENDING_BASE}/current"
LOG_DIR="${PMX_RESTORE_LOG_DIR:-/var/log/proxmenux}"
DEST_PREFIX="${PMX_RESTORE_DEST_PREFIX:-/}"
PRE_BACKUP_BASE="${PMX_RESTORE_PRE_BACKUP_BASE:-/root/proxmenux-pre-restore}"
RECOVERY_BASE="${PMX_RESTORE_RECOVERY_BASE:-/root/proxmenux-recovery}"
PRE_BACKUP_BASE="${PMX_RESTORE_PRE_BACKUP_BASE:-/var/lib/proxmenux/pre-restore}"
RECOVERY_BASE="${PMX_RESTORE_RECOVERY_BASE:-/var/lib/proxmenux/recovery}"
mkdir -p "$LOG_DIR" "$PENDING_BASE/completed" >/dev/null 2>&1 || true
LOG_FILE="${LOG_DIR}/proxmenux-restore-onboot-$(date +%Y%m%d_%H%M%S).log"
@@ -39,6 +39,9 @@ if [[ -f "$PLAN_ENV" ]]; then
fi
: "${HB_RESTORE_INCLUDE_ZFS:=0}"
: "${HB_COMPAT_CROSS_VERSION:=0}"
: "${HB_COMPAT_KERNEL_DIRECTION:=same}"
: "${HB_HYDRATION_APPLIED:=0}"
if [[ ! -f "$APPLY_LIST" ]]; then
echo "Apply list missing: $APPLY_LIST"
@@ -46,9 +49,30 @@ if [[ ! -f "$APPLY_LIST" ]]; then
exit 1
fi
echo "Pending dir: $PENDING_DIR"
echo "Apply list: $APPLY_LIST"
echo "Include ZFS: $HB_RESTORE_INCLUDE_ZFS"
echo "Pending dir: $PENDING_DIR"
echo "Apply list: $APPLY_LIST"
echo "Include ZFS: $HB_RESTORE_INCLUDE_ZFS"
echo "Cross-version: $HB_COMPAT_CROSS_VERSION"
echo "Kernel direction: $HB_COMPAT_KERNEL_DIRECTION"
# Hardware-drift skips persisted by _rs_prepare_pending_restore.
# Each line is an absolute path; we drop any rel path that matches
# (exact or descendant) before writing it back to disk. Without this
# the post-boot apply restored e.g. /etc/kernel/proxmox-boot-uuids
# from the backup, leaving the new bootloader pointing at a stale
# EFI UUID.
RS_SKIP_PATHS=""
SKIP_PATHS_FILE="$PENDING_DIR/rs-skip-paths.txt"
if [[ -f "$SKIP_PATHS_FILE" ]]; then
RS_SKIP_PATHS=$(cat "$SKIP_PATHS_FILE")
# Skips come from two sources today: hardware drift and
# cross-version safe restore. The label is generic so the log
# is truthful without having to reconcile the two upstream.
_skip_source_label="drift"
[[ "${HB_COMPAT_CROSS_VERSION:-0}" == "1" ]] && _skip_source_label="drift + cross-version"
echo "Skip paths: $(wc -l <"$SKIP_PATHS_FILE") entries (${_skip_source_label})"
fi
echo "running" >"$STATE_FILE"
backup_root="${PRE_BACKUP_BASE}/$(date +%Y%m%d_%H%M%S)-onboot"
@@ -70,7 +94,44 @@ while IFS= read -r rel; do
continue
fi
# Never restore cluster virtual filesystem data live.
# Hardware-drift skip filter — three cases:
# 1) "$rel" is exactly a skipped path → drop the whole rel
# 2) "$rel" is under a skipped path → drop the whole rel
# 3) A skipped path is under "$rel" (e.g. rel="etc/kernel" and
# skip="/etc/kernel/proxmox-boot-uuids") → keep applying rel
# but rsync with --exclude for the specific file, and NO
# --delete (so the host's existing file is preserved)
RSYNC_EXCLUDES=()
if [[ -n "$RS_SKIP_PATHS" ]]; then
_abs="/$rel"
_drop=0
while IFS= read -r _skip; do
[[ -z "$_skip" ]] && continue
if [[ "$_abs" == "$_skip" || "$_abs" == "$_skip"/* ]]; then
_drop=1
break
fi
# Case 3: skip is under our rel — build a relative exclude
if [[ "$_skip" == "$_abs"/* ]]; then
_rel_skip="${_skip#"$_abs"/}"
RSYNC_EXCLUDES+=(--exclude="$_rel_skip")
fi
done <<<"$RS_SKIP_PATHS"
if (( _drop )); then
echo " drift-skip: $rel"
((skipped++))
continue
fi
fi
# Cluster data (/etc/pve, /var/lib/pve-cluster) goes into a
# recovery dir for forensics/rollback, but unlike the live-
# menu apply path we ALSO apply it for real here: at this
# point in boot we're before networking.service, nothing is
# talking to the cluster yet, so a `systemctl stop pve-cluster`
# → copy → `systemctl start pve-cluster` is safe. This is the
# whole reason the operator picked "schedule remaining for
# next boot" instead of doing it live from SSH.
if [[ "$rel" == etc/pve* ]] || [[ "$rel" == var/lib/pve-cluster* ]]; then
if [[ -z "$cluster_recovery_root" ]]; then
cluster_recovery_root="${RECOVERY_BASE}/$(date +%Y%m%d_%H%M%S)-onboot"
@@ -78,6 +139,10 @@ while IFS= read -r rel; do
fi
mkdir -p "$cluster_recovery_root/$(dirname "$rel")" >/dev/null 2>&1 || true
cp -a "$src" "$cluster_recovery_root/$rel" >/dev/null 2>&1 || true
# Mark that we need to do the live apply at the end of
# the loop (we don't want to stop/start pve-cluster
# per-file — once is enough).
cluster_live_apply=1
((skipped++))
continue
fi
@@ -97,10 +162,21 @@ while IFS= read -r rel; do
if [[ -d "$src" ]]; then
mkdir -p "$dst" >/dev/null 2>&1 || true
if rsync -aAXH --delete "$src/" "$dst/" >/dev/null 2>&1; then
((applied++))
# When an exclude list is present, drop --delete so the host's
# copy of the excluded file isn't removed by rsync after being
# skipped from the source side.
if [[ ${#RSYNC_EXCLUDES[@]} -gt 0 ]]; then
if rsync -aAXH "${RSYNC_EXCLUDES[@]}" "$src/" "$dst/" >/dev/null 2>&1; then
((applied++))
else
((failed++))
fi
else
((failed++))
if rsync -aAXH --delete "$src/" "$dst/" >/dev/null 2>&1; then
((applied++))
else
((failed++))
fi
fi
else
mkdir -p "$(dirname "$dst")" >/dev/null 2>&1 || true
@@ -113,8 +189,13 @@ while IFS= read -r rel; do
done <"$APPLY_LIST"
systemctl daemon-reload >/dev/null 2>&1 || true
command -v update-initramfs >/dev/null 2>&1 && update-initramfs -u -k all >/dev/null 2>&1 || true
command -v update-grub >/dev/null 2>&1 && update-grub >/dev/null 2>&1 || true
# `update-initramfs -u -k all` and `update-grub` used to live here
# but: (a) they take 5-10 minutes for 3 kernels, hanging early-boot
# for that long, and (b) ifupdown2 was waiting on us. They now run
# AFTER pve-cluster is up via the apply_cluster_postboot.sh script
# we hook below, in the background where the user is already on the
# login prompt and using the system. Zero manual steps needed.
echo "Applied: $applied"
echo "Skipped: $skipped"
@@ -122,6 +203,8 @@ echo "Failed: $failed"
echo "Backup before restore: $backup_root"
if [[ -n "$cluster_recovery_root" ]]; then
# Always write the manual-helper script first — that's the
# rollback path if the live apply below blows up.
helper="${cluster_recovery_root}/apply-cluster-restore.sh"
cat > "$helper" <<EOF
#!/bin/bash
@@ -138,7 +221,17 @@ read -r -p "Type YES to continue: " ans
systemctl stop pve-cluster || true
[[ -d "\$RECOVERY_ROOT/etc/pve" ]] && mkdir -p /etc/pve && cp -a "\$RECOVERY_ROOT/etc/pve/." /etc/pve/ || true
[[ -d "\$RECOVERY_ROOT/var/lib/pve-cluster" ]] && mkdir -p /var/lib/pve-cluster && cp -a "\$RECOVERY_ROOT/var/lib/pve-cluster/." /var/lib/pve-cluster/ || true
if [[ -d "\$RECOVERY_ROOT/var/lib/pve-cluster" ]]; then
mkdir -p /var/lib/pve-cluster
cp -a "\$RECOVERY_ROOT/var/lib/pve-cluster/." /var/lib/pve-cluster/ || true
# If the backup only carried the raw-fallback (no sqlite3 dump),
# promote it to config.db so pve-cluster picks it up on start.
if [[ ! -f /var/lib/pve-cluster/config.db && -f /var/lib/pve-cluster/config.db.raw-fallback ]]; then
mv -f /var/lib/pve-cluster/config.db.raw-fallback /var/lib/pve-cluster/config.db
else
rm -f /var/lib/pve-cluster/config.db.raw-fallback
fi
fi
systemctl start pve-cluster || true
echo "Cluster recovery finished."
EOF
@@ -146,6 +239,108 @@ EOF
echo "Cluster paths extracted to: $cluster_recovery_root"
echo "Cluster recovery helper: $helper"
# We DON'T auto-apply /etc/pve here at boot because early-boot
# pve-cluster start blocks the unit (corosync etc. not ready).
# Instead we hand off to a SECOND oneshot unit that fires
# AFTER pve-cluster.service is up, when /etc/pve is the live
# pmxcfs FUSE mount and we can write individual files to it
# without restarting anything. That second unit is gated by
# ConditionPathExists on the marker file we drop here, so on
# a normal boot (no marker) it's a no-op.
if [[ "${cluster_live_apply:-0}" == "1" ]]; then
echo "Installing post-boot cluster apply unit..."
# Decide whether the post-boot script needs to run
# update-initramfs and/or update-grub by inspecting the
# apply list. Skipping them when nothing relevant was
# restored saves the operator 5-10 minutes of background
# initramfs rebuilds on EVERY restore — only do it when
# the backup actually touched paths that affect those
# tools' inputs.
NEEDS_INITRAMFS=0
NEEDS_GRUB=0
while IFS= read -r _rel; do
case "$_rel" in
etc/modules|etc/modules/*|\
etc/modules-load.d|etc/modules-load.d/*|\
etc/modprobe.d|etc/modprobe.d/*|\
etc/initramfs-tools|etc/initramfs-tools/*|\
etc/crypttab|\
etc/cryptsetup-initramfs|etc/cryptsetup-initramfs/*)
NEEDS_INITRAMFS=1 ;;
etc/default/grub|\
etc/kernel|etc/kernel/*|\
etc/grub.d|etc/grub.d/*)
NEEDS_GRUB=1 ;;
esac
done < "$APPLY_LIST"
# bk_older hydration writes tokens/modules/files DIRECTLY to
# the live target (bypassing the apply_list because the whole
# paths are in RS_SKIP_PATHS). Those changes still need
# update-initramfs + update-grub / proxmox-boot-tool refresh
# to take effect on the next boot — force the flags so the
# post-boot script picks them up.
if [[ "${HB_HYDRATION_APPLIED:-0}" == "1" ]]; then
NEEDS_INITRAMFS=1
NEEDS_GRUB=1
fi
echo "Post-boot maintenance flags: initramfs=$NEEDS_INITRAMFS grub=$NEEDS_GRUB"
# Marker as env-style key=value so the post-boot script
# can `source` it and read structured fields.
mkdir -p /var/lib/proxmenux >/dev/null 2>&1 || true
{
printf 'RECOVERY_ROOT=%s\n' "$cluster_recovery_root"
printf 'PENDING_DIR=%s\n' "$PENDING_DIR"
printf 'NEEDS_INITRAMFS=%s\n' "$NEEDS_INITRAMFS"
printf 'NEEDS_GRUB=%s\n' "$NEEDS_GRUB"
printf 'HB_COMPAT_CROSS_VERSION=%s\n' "${HB_COMPAT_CROSS_VERSION:-0}"
} > /var/lib/proxmenux/cluster-apply-pending
chmod 600 /var/lib/proxmenux/cluster-apply-pending
# Install the systemd unit. Idempotent: overwrite if it
# already exists (so script changes get picked up).
cat > /etc/systemd/system/proxmenux-apply-cluster-postboot.service <<UNITEOF
[Unit]
Description=ProxMenux Apply Cluster Configs (post-boot)
After=pve-cluster.service pveproxy.service network-online.target
Wants=pve-cluster.service
# Only fire on boots where pending_restore left us a marker.
# On every other boot, the condition fails and systemd skips
# us — zero overhead.
ConditionPathExists=/var/lib/proxmenux/cluster-apply-pending
[Service]
Type=oneshot
ExecStart=/usr/local/share/proxmenux/scripts/backup_restore/apply_cluster_postboot.sh
# Cap sized for the worst case: update-initramfs across all
# kernels + update-grub + cluster apply + component auto-reinstall
# (NVIDIA driver download + kernel-module build is the long pole).
# Service runs after pve-cluster is up, so this is purely background.
TimeoutStartSec=3600
[Install]
WantedBy=multi-user.target
UNITEOF
systemctl daemon-reload >/dev/null 2>&1 || true
systemctl enable proxmenux-apply-cluster-postboot.service >/dev/null 2>&1 || true
# `systemctl enable` only adds the unit to multi-user.target.wants/.
# It does NOT pull the unit into the currently-running boot
# transaction — by the time we run, multi-user.target may have
# already collected its wants. `start --no-block` schedules the
# unit for activation respecting its After= ordering (pve-cluster
# comes up first), without blocking apply_pending_restore.sh
# itself. Without this, the postboot unit only fires on the
# NEXT reboot, defeating the "single reboot, zero manual steps"
# promise.
systemctl start --no-block proxmenux-apply-cluster-postboot.service >/dev/null 2>&1 || true
echo "Cluster apply will run automatically after pve-cluster comes up."
echo "Fallback manual: bash $helper"
fi
fi
if [[ "$failed" -eq 0 ]]; then
+2810 -395
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+438 -36
View File
@@ -75,13 +75,63 @@ _list_jobs() {
done | sort
}
# Same as _list_jobs but skips one-shot manual runs (Sprint B).
# Manual entries are `.env` files with `MANUAL_RUN=1` — they're
# closed executions, not configured tasks. Used by "Show jobs"
# and "Run a job now" so those views only surface real backup
# tasks; Delete / Toggle keep using _list_jobs so the operator
# can still clean up leftover manual entries from those menus.
_list_scheduled_jobs() {
local f id
for f in "$JOBS_DIR"/*.env; do
[[ -f "$f" ]] || continue
grep -q '^MANUAL_RUN=1$' "$f" 2>/dev/null && continue
id=$(basename "$f" .env)
printf '%s\n' "$id"
done | sort
}
# Returns 0 if the job is attached to a PVE vzdump storage (no systemd
# timer — the trigger comes from the vzdump hook, matched by PVE_STORAGE
# against $STOREID set by PVE for every backup phase).
_job_is_attached() {
local id="$1" f
f=$(_job_file "$id")
[[ -f "$f" ]] || return 1
grep -q "^PVE_STORAGE=" "$f"
}
# Reads a key=val pair from the job .env file (handles `printf %q`
# quoting that _write_job_env produces).
_job_env_get() {
local id="$1" key="$2" f raw
f=$(_job_file "$id")
[[ -f "$f" ]] || return 1
raw=$(grep -E "^${key}=" "$f" | head -1 | cut -d'=' -f2-)
eval "echo $raw" 2>/dev/null || echo "$raw"
}
_show_job_status() {
local id="$1"
local timer_state="disabled"
local service_state="unknown"
if _job_is_attached "$id"; then
local storage
storage=$(_job_env_get "$id" "PVE_STORAGE")
local enabled
enabled=$(_job_env_get "$id" "ENABLED")
[[ "$enabled" == "0" ]] && { echo "attached(disabled) → storage:$storage"; return; }
echo "attached → storage:$storage"
return
fi
local timer_state="disabled" service_state
systemctl is-enabled --quiet "proxmenux-backup-${id}.timer" >/dev/null 2>&1 && timer_state="enabled"
service_state=$(systemctl is-active "proxmenux-backup-${id}.service" 2>/dev/null || echo "inactive")
echo "${timer_state}/${service_state}"
if [[ "$service_state" == "active" ]]; then
echo "running"
elif [[ "$timer_state" == "enabled" ]]; then
echo "enabled"
else
echo "disabled"
fi
}
_write_job_units() {
@@ -155,10 +205,146 @@ _prompt_retention() {
)
}
# Builds a "host backup attached to a PVE vzdump job" — no systemd
# timer is created; the trigger is the vzdump hook that fires when
# the parent job runs. Schedule and retention come from the parent.
_create_job_attached() {
local id="$1"
local backend="$2"
local -a jobs=()
mapfile -t jobs < <(hb_pve_list_vzdump_jobs_for_backend "$backend")
if (( ${#jobs[@]} == 0 )); then
dialog --backtitle "ProxMenux" --title "$(translate "No compatible PVE jobs")" \
--msgbox "$(translate "No PVE vzdump job uses a") $backend $(translate "storage.")" 8 70
return 1
fi
local -a menu=()
local i=1 row pve_id pve_storage _ pve_schedule _pve_prune pve_enabled
for row in "${jobs[@]}"; do
IFS=$'\t' read -r pve_id pve_storage _ pve_schedule _pve_prune pve_enabled <<<"$row"
local label="${pve_id} · ${pve_storage} · ${pve_schedule}"
[[ "$pve_enabled" == "0" ]] && label+=" $(translate "(disabled)")"
menu+=("$i" "$label")
((i++))
done
local sel
sel=$(dialog --backtitle "ProxMenux" --title "$(translate "Pick PVE vzdump job")" \
--menu "\n$(translate "Select the parent job to attach to:")" \
"$HB_UI_MENU_H" "$HB_UI_MENU_W" "$HB_UI_MENU_LIST" "${menu[@]}" \
3>&1 1>&2 2>&3) || return 1
local pve_prune
IFS=$'\t' read -r pve_id pve_storage _ pve_schedule pve_prune pve_enabled <<<"${jobs[$((sel-1))]}"
local profile_mode
profile_mode=$(dialog --backtitle "ProxMenux" --title "$(translate "Profile")" \
--menu "\n$(translate "Select backup profile:")" 12 68 4 \
"default" "Default critical paths" \
"custom" "Custom selected paths" \
3>&1 1>&2 2>&3) || return 1
local -a paths=()
hb_select_profile_paths "$profile_mode" paths || return 1
local -a lines=(
"JOB_ID=$id"
"BACKEND=$backend"
"PVE_PARENT_JOB=$pve_id"
"PVE_STORAGE=$pve_storage"
"PROFILE_MODE=$profile_mode"
"ENABLED=1"
)
# Inherit retention from the parent job (one KEEP_* per prune-backups key).
local kv
while IFS= read -r kv; do
[[ -n "$kv" ]] && lines+=("$kv")
done < <(hb_pve_prune_to_keep_env "$pve_prune")
case "$backend" in
pbs)
hb_select_pbs_repository || return 1
# Attached-mode PBS jobs were being written without ever asking
# about encryption — the runner then invoked
# `proxmox-backup-client backup` with no `--keyfile`, so every
# attached-mode backup landed on PBS unencrypted regardless of
# what the operator would have picked. Same encryption prompt as
# `_create_job_new`: hb_ask_pbs_encryption sets
# HB_PBS_KEYFILE_OPT / HB_PBS_ENC_PASS which we translate into
# PBS_KEYFILE / PBS_ENCRYPTION_PASSWORD env lines the runner
# reads. Cancel from the encryption dialog aborts the wizard
# (return 1) so no half-configured job is saved.
hb_ask_pbs_encryption || return 1
local bid
bid="hostcfg-$(hostname)"
bid=$(dialog --backtitle "ProxMenux" --title "PBS" \
--inputbox "$(translate "Backup ID for this job:")" \
"$HB_UI_INPUT_H" "$HB_UI_INPUT_W" "$bid" 3>&1 1>&2 2>&3) || return 1
bid=$(echo "$bid" | tr -cs '[:alnum:]_-' '-' | sed 's/-*$//')
# Same derivation as _create_job_new: HB_PBS_KEYFILE_OPT is the
# full "--keyfile /path" string when encryption is accepted,
# empty otherwise. Non-empty → write PBS_KEYFILE to the canonical
# path so the Monitor Web (which detects encryption from a
# non-empty PBS_KEYFILE line) shows the job as encrypted.
local pbs_kf_val=""
[[ -n "${HB_PBS_KEYFILE_OPT:-}" ]] && pbs_kf_val="$HB_STATE_DIR/pbs-key.conf"
lines+=(
"PBS_REPOSITORY=${HB_PBS_REPOSITORY}"
"PBS_PASSWORD=${HB_PBS_SECRET}"
"PBS_BACKUP_ID=${bid}"
"PBS_KEYFILE=${pbs_kf_val}"
"PBS_ENCRYPTION_PASSWORD=${HB_PBS_ENC_PASS:-}"
# Resolved by hb_select_pbs_repository. Persist it so the runner
# doesn't have to re-derive it — it can only do that for
# Datacenter-managed storages.
"PBS_FINGERPRINT=${HB_PBS_FINGERPRINT:-}"
)
;;
local)
# Derive the dump directory from the storage entry. PVE stores
# vzdump archives under <path>/dump/ when the storage is dir/nfs.
local dest_dir="/var/lib/vz/dump"
local sp
sp=$(awk -v sid="$pve_storage" '
/^[a-z]+:[[:space:]]/ { in_block=($2==sid) }
in_block && /^[[:space:]]+path[[:space:]]/ { sub(/^[[:space:]]+path[[:space:]]+/,""); print; exit }
' /etc/pve/storage.cfg) || true
[[ -n "$sp" ]] && dest_dir="${sp%/}/dump"
lines+=("LOCAL_DEST_DIR=$dest_dir" "LOCAL_ARCHIVE_EXT=tar.zst")
;;
esac
_write_job_env "$(_job_file "$id")" "${lines[@]}"
: > "$(_job_paths_file "$id")"
local p
for p in "${paths[@]}"; do
echo "$p" >> "$(_job_paths_file "$id")"
done
# No unit / timer — the trigger is the vzdump hook fired by the parent PVE job.
hb_install_vzdump_hook >/dev/null 2>&1 || \
msg_warn "$(translate "Could not install vzdump hook in /etc/vzdump.conf")"
show_proxmenux_logo
msg_title "$(translate "Host backup attached to PVE job")"
echo
echo -e "${TAB}${BGN}$(translate "Job ID:")${CL} ${BL}${id}${CL}"
echo -e "${TAB}${BGN}$(translate "Attached to PVE job:")${CL} ${BL}${pve_id}${CL}"
echo -e "${TAB}${BGN}$(translate "Inherited schedule:")${CL} ${BL}${pve_schedule}${CL}"
echo -e "${TAB}${BGN}$(translate "Inherited retention:")${CL} ${BL}${pve_prune}${CL}"
echo -e "${TAB}${BGN}$(translate "Backend:")${CL} ${BL}${backend}${pve_storage}${CL}"
echo
msg_success "$(translate "Press Enter to continue...")"
read -r
return 0
}
_create_job() {
local id backend on_calendar profile_mode
id=$(dialog --backtitle "ProxMenux" --title "$(translate "New backup job")" \
--inputbox "$(translate "Job ID (letters, numbers, - _)")" 9 68 "hostcfg-daily" 3>&1 1>&2 2>&3) || return 1
--inputbox "$(translate "Job ID (letters, numbers, - _)")" 9 68 "my-host-backup" 3>&1 1>&2 2>&3) || return 1
[[ -z "$id" ]] && return 1
id=$(echo "$id" | tr -cs '[:alnum:]_-' '-' | sed 's/^-*//; s/-*$//')
[[ -z "$id" ]] && return 1
@@ -168,13 +354,42 @@ _create_job() {
return 1
}
# Order: recommended backend first (PBS gets chunk-based dedup, native
# PVE integration and paired keyfile recovery), then Borg, then plain
# local archive. Matches the "recommended → alternatives" ordering used
# elsewhere in the UI so a first-time user lands on the best default.
backend=$(dialog --backtitle "ProxMenux" --title "$(translate "Backend")" \
--menu "\n$(translate "Select backup backend:")" 14 70 6 \
"local" "Local archive" \
"pbs" "Proxmox Backup Server (recommended)" \
"borg" "Borg repository" \
"pbs" "Proxmox Backup Server" \
"local" "Local archive" \
3>&1 1>&2 2>&3) || return 1
# Offer attach-mode for backends that map to a PVE storage. The
# vzdump scheduler in PVE already handles trigger + retention for
# VM/CT backups; the host backup can ride alongside it via a hook.
# Borg has no PVE-side scheduler, so attach makes no sense there.
if [[ "$backend" == "pbs" || "$backend" == "local" ]]; then
local creation_mode
creation_mode=$(dialog --backtitle "ProxMenux" --title "$(translate "How to schedule")" \
--menu "\n$(translate "Choose how this host backup will be triggered:")" 14 78 4 \
"new" "$(translate "New scheduled job (own timer + retention)")" \
"attach" "$(translate "Attach to an existing PVE vzdump job (inherit schedule + retention)")" \
3>&1 1>&2 2>&3) || return 1
if [[ "$creation_mode" == "attach" ]]; then
# If no compatible PVE job exists yet, show a helpful pointer
# instead of silently dropping back to "new" mode.
if [[ -z "$(hb_pve_list_vzdump_jobs_for_backend "$backend" 2>/dev/null | head -1)" ]]; then
dialog --backtitle "ProxMenux" --title "$(translate "No compatible PVE jobs")" \
--msgbox "$(translate "No PVE vzdump job uses a") $backend $(translate "storage yet.")"$'\n\n'"$(translate "Create one first in Datacenter → Backup, then return here to attach.")" \
12 78
return 1
fi
_create_job_attached "$id" "$backend"
return $?
fi
fi
on_calendar=$(dialog --backtitle "ProxMenux" --title "$(translate "Schedule")" \
--inputbox "$(translate "systemd OnCalendar expression")"$'\n'"$(translate "Example: daily or Mon..Fri 03:00")" \
11 72 "daily" 3>&1 1>&2 2>&3) || return 1
@@ -204,7 +419,7 @@ _create_job() {
case "$backend" in
local)
local dest_dir ext
dest_dir=$(hb_prompt_dest_dir) || return 1
dest_dir=$(hb_select_local_target) || return 1
ext=$(dialog --backtitle "ProxMenux" --title "$(translate "Archive format")" \
--menu "\n$(translate "Select local archive format:")" 12 62 4 \
"tar.zst" "tar + zstd (preferred)" \
@@ -225,19 +440,37 @@ _create_job() {
;;
pbs)
hb_select_pbs_repository || return 1
hb_ask_pbs_encryption
# Propagate the operator cancel from the encryption dialog so the
# wizard drops back to the previous step instead of saving a job
# spec with a half-configured encryption block.
hb_ask_pbs_encryption || return 1
local bid
bid="hostcfg-$(hostname)"
bid=$(dialog --backtitle "ProxMenux" --title "PBS" \
--inputbox "$(translate "Backup ID for this job:")" \
"$HB_UI_INPUT_H" "$HB_UI_INPUT_W" "$bid" 3>&1 1>&2 2>&3) || return 1
bid=$(echo "$bid" | tr -cs '[:alnum:]_-' '-' | sed 's/-*$//')
# PBS_KEYFILE: `hb_ask_pbs_encryption` sets HB_PBS_KEYFILE_OPT
# to the FULL CLI flag string ("--keyfile /path") when
# encryption is accepted, and leaves it empty otherwise. The
# previous read of `HB_PBS_KEYFILE` was picking up an unset
# variable and always wrote an empty PBS_KEYFILE — so the
# Monitor Web (which detects encryption from a non-empty
# PBS_KEYFILE line) surfaced every CLI-created encrypted job
# as unencrypted. Derive the canonical path from
# HB_PBS_KEYFILE_OPT presence instead.
local pbs_kf_val=""
[[ -n "${HB_PBS_KEYFILE_OPT:-}" ]] && pbs_kf_val="$HB_STATE_DIR/pbs-key.conf"
lines+=(
"PBS_REPOSITORY=${HB_PBS_REPOSITORY}"
"PBS_PASSWORD=${HB_PBS_SECRET}"
"PBS_BACKUP_ID=${bid}"
"PBS_KEYFILE=${HB_PBS_KEYFILE:-}"
"PBS_KEYFILE=${pbs_kf_val}"
"PBS_ENCRYPTION_PASSWORD=${HB_PBS_ENC_PASS:-}"
# Resolved by hb_select_pbs_repository. Persist it so the runner
# doesn't have to re-derive it — it can only do that for
# Datacenter-managed storages.
"PBS_FINGERPRINT=${HB_PBS_FINGERPRINT:-}"
)
;;
esac
@@ -269,19 +502,37 @@ _create_job() {
_pick_job() {
local title="$1"
local __out_var="$2"
# Optional scope: "all" (default) surfaces every .env in JOBS_DIR
# including one-shot manual runs; "scheduled" filters those out so
# Run-now only shows real configured tasks. Delete / Toggle keep
# the default so the operator can still clean up manual leftovers.
local scope="${3:-all}"
local -a ids=()
mapfile -t ids < <(_list_jobs)
if [[ "$scope" == "scheduled" ]]; then
mapfile -t ids < <(_list_scheduled_jobs)
else
mapfile -t ids < <(_list_jobs)
fi
if [[ ${#ids[@]} -eq 0 ]]; then
dialog --backtitle "ProxMenux" --title "$(translate "No jobs")" \
--msgbox "$(translate "No scheduled backup jobs found.")" 8 62
return 1
fi
# Build the menu rows. The loop variable is INTENTIONALLY named
# `_iter_id` (not `id`) — every caller passes "id" as $__out_var so
# the nameref below should point at the caller's local. A loop
# variable named `id` here would shadow it, and the nameref would
# silently write into _pick_job's own scope instead, leaving the
# caller with an empty string. That manifested as:
# ✓ Job timer enabled: (empty)
# run_scheduled_backup.sh: Usage: ... <job_id>
# Both reported on 2026-06-07.
local -a menu=()
local i=1 id
for id in "${ids[@]}"; do
menu+=("$i" "$id [$(_show_job_status "$id")]")
local i=1 _iter_id
for _iter_id in "${ids[@]}"; do
menu+=("$i" "$_iter_id [$(_show_job_status "$_iter_id")]")
((i++))
done
local sel
@@ -295,29 +546,101 @@ _pick_job() {
return 0
}
# Common screen reset for any post-dialog action result. The
# `dialog` calls in this script leave their box drawn on screen
# even after the user has confirmed; without this reset, the
# subsequent msg_ok / msg_warn / "Press Enter" output renders
# in the bottom-left corner UNDER the leftover dialog box.
# show_proxmenux_logo already runs `clear` internally, so we
# don't add another one — the convention used across proxmenux
# (create_vm_menu.sh, config_menu.sh, menu_post_install.sh) is:
# show_proxmenux_logo → msg_title → result message
# Reported 2026-06-07 when the operator hit "Run job now" and
# saw "Job executed successfully" floating over the picker.
_render_action_screen() {
show_proxmenux_logo
msg_title "$1"
}
_job_run_now() {
# "scheduled" scope — one-shot manual runs are closed executions,
# not tasks to re-fire. Filtering them out of this picker prevents
# accidental re-runs of an operator's past manual backups.
local id=""
_pick_job "$(translate "Run job now")" id || return 1
_pick_job "$(translate "Run job now")" id scheduled || return 1
# Defensive guard against a future regression of the nameref-shadowing
# bug that left $id empty here on 2026-06-07. Without this, the runner
# gets called with no argument and emits "Usage: ... <job_id>".
if [[ -z "$id" ]]; then
_render_action_screen "$(translate "Run job now")"
msg_error "$(translate "Job selection returned empty id — aborting.")"
msg_success "$(translate "Press Enter to continue...")"
read -r
return 1
fi
local runner="$LOCAL_SCRIPTS/backup_restore/run_scheduled_backup.sh"
[[ ! -f "$runner" ]] && runner="$SCRIPT_DIR/run_scheduled_backup.sh"
if "$runner" "$id"; then
msg_ok "$(translate "Job executed successfully.")"
else
msg_warn "$(translate "Job execution finished with errors. Check logs.")"
fi
# Foreground execution — the runner detects TTY and prints a
# colored progress layout (mirrors _bk_local in backup_host.sh).
# Plain-text log file is still written for audit / scheduler runs.
_render_action_screen "$(translate "Running backup job:") $id"
echo
"$runner" "$id"
local runner_exit
runner_exit=$?
echo
msg_success "$(translate "Press Enter to continue...")"
read -r
return $runner_exit
}
_job_toggle() {
# "scheduled" scope — one-shot manual runs carry ENABLED=0 by
# definition and can't be re-fired from a timer, so offering them
# in this picker was noise. Delete keeps the "all" scope so the
# operator can still clean up manual entries from that menu.
local id=""
_pick_job "$(translate "Enable/Disable job")" id || return 1
if systemctl is-enabled --quiet "proxmenux-backup-${id}.timer" >/dev/null 2>&1; then
systemctl disable --now "proxmenux-backup-${id}.timer" >/dev/null 2>&1 || true
msg_warn "$(translate "Job timer disabled:") $id"
_pick_job "$(translate "Enable/Disable job")" id scheduled || return 1
if [[ -z "$id" ]]; then
_render_action_screen "$(translate "Enable/Disable job")"
msg_error "$(translate "Job selection returned empty id — aborting.")"
msg_success "$(translate "Press Enter to continue...")"
read -r
return 1
fi
local action_label
if _job_is_attached "$id"; then
# Attached jobs have no systemd timer — flip the ENABLED flag in
# the .env so the vzdump hook respects it on the next parent run.
local f current
f=$(_job_file "$id")
current=$(_job_env_get "$id" "ENABLED")
if [[ "$current" == "0" ]]; then
sed -i 's/^ENABLED=.*/ENABLED=1/' "$f"
action_label="enabled"
else
sed -i 's/^ENABLED=.*/ENABLED=0/' "$f"
action_label="disabled"
fi
else
systemctl enable --now "proxmenux-backup-${id}.timer" >/dev/null 2>&1 || true
msg_ok "$(translate "Job timer enabled:") $id"
if systemctl is-enabled --quiet "proxmenux-backup-${id}.timer" >/dev/null 2>&1; then
systemctl disable --now "proxmenux-backup-${id}.timer" >/dev/null 2>&1 || true
action_label="disabled"
else
systemctl enable --now "proxmenux-backup-${id}.timer" >/dev/null 2>&1 || true
action_label="enabled"
fi
fi
_render_action_screen "$(translate "Enable/Disable job")"
if [[ "$action_label" == "disabled" ]]; then
msg_warn "$(translate "Job disabled:") $id"
else
msg_ok "$(translate "Job enabled:") $id"
fi
msg_success "$(translate "Press Enter to continue...")"
read -r
@@ -326,13 +649,34 @@ _job_toggle() {
_job_delete() {
local id=""
_pick_job "$(translate "Delete job")" id || return 1
# An empty id here would build malformed unit paths like
# /etc/systemd/system/proxmenux-backup-.timer, and the subsequent
# rm -f would silently no-op against bogus paths — making it LOOK
# like a successful delete while the real job stays untouched.
if [[ -z "$id" ]]; then
_render_action_screen "$(translate "Delete job")"
msg_error "$(translate "Job selection returned empty id — aborting.")"
msg_success "$(translate "Press Enter to continue...")"
read -r
return 1
fi
local confirm_body
confirm_body="$(translate "Delete scheduled backup job?")"$'\n\n'"ID: ${id}"
if _job_is_attached "$id"; then
local storage
storage=$(_job_env_get "$id" "PVE_STORAGE")
confirm_body+=$'\n'"$(translate "Type: attached to PVE storage") ${storage}"
confirm_body+=$'\n\n'"$(translate "Only the host backup hook is removed — PVE vzdump jobs targeting this storage stay intact.")"
fi
if ! whiptail --title "$(translate "Confirm delete")" \
--yesno "$(translate "Delete scheduled backup job?")"$'\n\n'"ID: ${id}" 10 66; then
--yesno "$confirm_body" 14 70; then
return 1
fi
systemctl disable --now "proxmenux-backup-${id}.timer" >/dev/null 2>&1 || true
rm -f "$(_service_file "$id")" "$(_timer_file "$id")" "$(_job_file "$id")" "$(_job_paths_file "$id")"
systemctl daemon-reload >/dev/null 2>&1 || true
_render_action_screen "$(translate "Delete job")"
msg_ok "$(translate "Job deleted:") $id"
msg_success "$(translate "Press Enter to continue...")"
read -r
@@ -341,21 +685,79 @@ _job_delete() {
_show_jobs() {
local tmp
tmp=$(mktemp) || return
# Per-job block: header line with status badge + 3 detail lines
# summarising the backend destination and the last run — same
# information the Monitor UI shows on the Backups tab, condensed
# for the shell dialog.
local -a job_ids=()
local id
while IFS= read -r id; do
[[ -z "$id" ]] && continue
job_ids+=("$id")
done < <(_list_scheduled_jobs)
{
echo "=== $(translate "Scheduled backup jobs") ==="
echo ""
local id
while IFS= read -r id; do
[[ -z "$id" ]] && continue
echo "$id [$(_show_job_status "$id")]"
if [[ -f "${LOG_DIR}/${id}-last.status" ]]; then
sed 's/^/ /' "${LOG_DIR}/${id}-last.status"
fi
echo ""
done < <(_list_jobs)
if [[ ${#job_ids[@]} -eq 0 ]]; then
translate "No scheduled backup jobs configured."
else
local status backend dest label profile last_run last_result
for id in "${job_ids[@]}"; do
status=$(_show_job_status "$id")
backend=$(_job_env_get "$id" BACKEND || echo "")
profile=$(_job_env_get "$id" PROFILE_MODE || echo default)
case "$backend" in
pbs)
local pbs_repo pbs_bid
pbs_repo=$(_job_env_get "$id" PBS_REPOSITORY || echo "?")
pbs_bid=$(_job_env_get "$id" PBS_BACKUP_ID || echo "?")
dest="$pbs_repo (id=$pbs_bid)"
label="PBS"
;;
local)
dest=$(_job_env_get "$id" LOCAL_DEST_DIR || echo "?")
label="Local archive"
;;
borg)
dest=$(_job_env_get "$id" BORG_REPO || echo "?")
label="Borg"
;;
*)
dest="?"
label="${backend:-?}"
;;
esac
last_run=""; last_result=""
if [[ -f "${LOG_DIR}/${id}-last.status" ]]; then
local last_at
last_at=$(grep -m1 '^RUN_AT=' "${LOG_DIR}/${id}-last.status" | cut -d= -f2-)
last_result=$(grep -m1 '^RESULT=' "${LOG_DIR}/${id}-last.status" | cut -d= -f2-)
# Trim the timezone suffix for readability (14:13:54 vs 14:13:54+02:00)
last_run="${last_at%%+*}"
last_run="${last_run%%-[0-9][0-9]:[0-9][0-9]}"
last_run="${last_run/T/ }"
fi
printf "• %s [%s]\n" "$id" "$status"
printf " %-9s %s · %s\n" "$(translate "Backend:")" "$label" "$dest"
printf " %-9s %s\n" "$(translate "Profile:")" "$profile"
if [[ -n "$last_run" ]]; then
printf " %-9s %s → %s\n" "$(translate "Last run:")" "$last_run" "${last_result:-?}"
else
printf " %-9s %s\n" "$(translate "Last run:")" "$(translate "never")"
fi
echo ""
done
fi
} > "$tmp"
# Same window size as the parent scheduler menu — keeps the
# dimensions consistent across the flow instead of one dialog
# shrinking or growing per view.
dialog --backtitle "ProxMenux" --title "$(translate "Scheduled backup jobs")" \
--textbox "$tmp" 28 100 || true
--textbox "$tmp" "$HB_UI_MENU_H" "$HB_UI_MENU_W" || true
rm -f "$tmp"
}
+126
View File
@@ -0,0 +1,126 @@
#!/usr/bin/env bash
# ==========================================================
# ProxMenux backup manifest orchestrator
# ==========================================================
# Composes the six collectors into one manifest.json that
# validates against schema/manifest.schema.json. Designed to
# be called by backup_host.sh during a backup run. Read-only
# (no side effects on the host).
#
# Usage:
# build_manifest.sh [--paths-archived <path1> <path2> ...]
# build_manifest.sh --validate (re-runs the JSON Schema validation)
#
# Stdout: pretty-printed manifest JSON.
# Stderr: progress + warnings.
# ==========================================================
set -euo pipefail
COLLECTORS_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCHEMA_FILE="$COLLECTORS_DIR/../schema/manifest.schema.json"
# Parse flags
paths_archived='null'
do_validate=0
while [[ $# -gt 0 ]]; do
case "$1" in
--paths-archived)
shift
tmp='[]'
while [[ $# -gt 0 && "$1" != --* ]]; do
tmp="$(jq --argjson a "$tmp" --arg p "$1" -n '$a + [$p]')"
shift
done
paths_archived="$tmp"
;;
--validate)
do_validate=1; shift ;;
-h|--help)
sed -nE '/^# Usage:/,/^# Stderr:/p' "$0" | sed -E 's/^# ?//' >&2
exit 0
;;
*) shift ;;
esac
done
# Run each collector. If a collector fails we fall back to a safe default
# (empty array / null object) and warn — the manifest is still useful even
# if one section is incomplete.
run_collector() {
local name="$1" fallback="$2"
local out
if out="$(bash "$COLLECTORS_DIR/$name" 2>>/tmp/proxmenux-manifest-stderr.log)"; then
printf '%s' "$out"
else
printf 'warning: collector %s failed; using fallback\n' "$name" >&2
printf '%s' "$fallback"
fi
}
# Empty error log first so we can attribute failures to this run.
: >/tmp/proxmenux-manifest-stderr.log
source_host="$(run_collector collect_source_host.sh '{}')"
hardware_inventory="$(run_collector collect_hardware.sh '{"gpu":[],"tpu":[],"nic":[],"wireless":[]}')"
storage_inventory="$(run_collector collect_storage.sh '{"zfs_pools":[],"lvm":{"vgs":[]},"physical_disks":[],"pve_storage_cfg":[],"mounts":[]}')"
installed_components="$(run_collector collect_proxmenux_state.sh '[]')"
kernel_params="$(run_collector collect_kernel.sh '{"cmdline_extra":[],"modules_loaded_at_boot":[],"modprobe_d_files":[]}')"
guests="$(run_collector collect_guests.sh '{"vms":[],"lxcs":[]}')"
created_at="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
# Compose the final manifest. The wrapper key matches the schema:
# the top level is a single "proxmenux_backup_manifest" object.
manifest="$(jq -n \
--arg created_at "$created_at" \
--arg created_by "proxmenux-host-backup/1.3.0" \
--argjson source_host "$source_host" \
--argjson hardware "$hardware_inventory" \
--argjson storage "$storage_inventory" \
--argjson components "$installed_components" \
--argjson kernel "$kernel_params" \
--argjson guests "$guests" \
--argjson paths_archived "$paths_archived" \
'{
proxmenux_backup_manifest: {
schema_version: 1,
created_at: $created_at,
created_by: $created_by,
source_host: $source_host,
hardware_inventory: $hardware,
storage_inventory: $storage,
proxmenux_installed_components: $components,
kernel_params: $kernel,
vms_lxcs_at_backup: $guests,
backup_metadata: {
encrypted: false,
encryption_format: null,
compression: "zstd",
paths_archived: $paths_archived,
sha256_archive: null,
size_bytes: null
}
}
}')"
# Optional validation step. If python3 + jsonschema are available, run
# them; otherwise silently skip (validation is mostly a developer aid).
if [[ "$do_validate" == 1 ]]; then
if command -v python3 >/dev/null 2>&1 && python3 -c 'import jsonschema' 2>/dev/null; then
printf '%s' "$manifest" | python3 -c "
import json, sys, jsonschema
schema = json.load(open('$SCHEMA_FILE'))
inst = json.load(sys.stdin)
try:
jsonschema.validate(instance=inst, schema=schema)
print('manifest: validates against schema', file=sys.stderr)
except jsonschema.exceptions.ValidationError as e:
print(f'manifest: SCHEMA VIOLATION at {list(e.absolute_path)}: {e.message}', file=sys.stderr)
sys.exit(1)
"
else
printf 'manifest: jsonschema python module not present; skipping validation\n' >&2
fi
fi
printf '%s\n' "$manifest"
+97
View File
@@ -0,0 +1,97 @@
#!/usr/bin/env bash
# ==========================================================
# ProxMenux backup manifest collector — vms_lxcs_at_backup
# ==========================================================
# Enumerates VMs (qm list) and LXCs (pct list) on this PVE node.
# Read-only; emits the metadata only — actual VM/LXC data is
# the responsibility of vzdump / PBS, not this manifest.
# Schema: scripts/backup_restore/schema/manifest.schema.json
# ==========================================================
set -euo pipefail
vms='[]'
lxcs='[]'
# ── VMs (qm list) ──
# Output:
# VMID NAME STATUS MEM(MB) BOOTDISK(GB) PID
# 100 Alpine-Linux-3-21 stopped 4096 0.00 0
# Header line starts with VMID; we skip it.
if command -v qm >/dev/null 2>&1; then
while IFS= read -r line; do
[[ -z "$line" ]] && continue
# Skip the header
[[ "$line" =~ ^[[:space:]]*VMID[[:space:]] ]] && continue
# Parse positionally. NAME can contain spaces, but `qm list` pads/columns
# them, so we use fixed positions: VMID at col 1, STATUS as the 3rd
# whitespace-delimited token from the END (mem, bootdisk, pid are after).
vmid="$(printf '%s' "$line" | awk '{print $1}')"
[[ "$vmid" =~ ^[0-9]+$ ]] || continue
# Strip trailing PID + BOOTDISK + MEM(MB) + STATUS to extract the NAME.
# rev → cut → rev technique:
trailing="$(printf '%s' "$line" | awk '{printf "%s %s %s %s", $(NF-3), $(NF-2), $(NF-1), $NF}')"
status="$(printf '%s' "$trailing" | awk '{print $1}')"
memory_mb="$(printf '%s' "$trailing" | awk '{print $2}')"
bootdisk_gb="$(printf '%s' "$trailing" | awk '{print $3}')"
# Name: drop first column (vmid) and last 4 columns
name="$(printf '%s' "$line" | awk '{$1=""; for(i=NF-3;i<=NF;i++) $i=""; sub(/^[[:space:]]+/,""); sub(/[[:space:]]+$/,""); print}')"
case "$status" in
running|stopped|paused) ;;
*) status="stopped" ;;
esac
vms="$(jq --argjson acc "$vms" \
--argjson vmid "$vmid" \
--arg name "$name" \
--argjson memory_mb "${memory_mb:-0}" \
--argjson bootdisk_gb "${bootdisk_gb:-0}" \
--arg status "$status" \
-n '
$acc + [{
vmid: $vmid,
name: $name,
memory_mb: $memory_mb,
bootdisk_gb: $bootdisk_gb,
status: $status,
config_file: ("configs/qemu-server/" + ($vmid|tostring) + ".conf")
}]
')"
done < <(qm list 2>/dev/null || true)
fi
# ── LXCs (pct list) ──
# Output:
# VMID Status Lock Name
# 101 running alpine
# Header line starts with VMID; we skip it.
if command -v pct >/dev/null 2>&1; then
while IFS= read -r line; do
[[ -z "$line" ]] && continue
[[ "$line" =~ ^[[:space:]]*VMID[[:space:]] ]] && continue
vmid="$(printf '%s' "$line" | awk '{print $1}')"
[[ "$vmid" =~ ^[0-9]+$ ]] || continue
status="$(printf '%s' "$line" | awk '{print $2}')"
# Lock column is sparse; name is always last positional non-empty token
name="$(printf '%s' "$line" | awk '{print $NF}')"
case "$status" in
running|stopped) ;;
*) status="stopped" ;;
esac
lxcs="$(jq --argjson acc "$lxcs" \
--argjson vmid "$vmid" \
--arg name "$name" \
--arg status "$status" \
-n '
$acc + [{
vmid: $vmid,
name: $name,
status: $status,
config_file: ("configs/lxc/" + ($vmid|tostring) + ".conf")
}]
')"
done < <(pct list 2>/dev/null || true)
fi
jq -n --argjson vms "$vms" --argjson lxcs "$lxcs" \
'{ vms: $vms, lxcs: $lxcs }'
+222
View File
@@ -0,0 +1,222 @@
#!/usr/bin/env bash
# ==========================================================
# ProxMenux backup manifest collector — hardware_inventory
# ==========================================================
# Detects GPUs (with vendor → ProxMenux installer mapping),
# TPUs (Coral PCIe/USB), NICs (with bridge membership), and
# Wireless interfaces. Read-only. Schema:
# scripts/backup_restore/schema/manifest.schema.json
# ==========================================================
set -euo pipefail
# Vendor → installer path mapping. Update when ProxMenux adds new
# installers for hardware that depends on out-of-tree drivers.
# Vendors WITHOUT a mapping get null (e.g. Intel/AMD iGPUs work with
# in-tree drivers, no special installer needed).
gpu_installer_for() {
case "$1" in
NVIDIA) echo "scripts/gpu_tpu/nvidia_installer.sh" ;;
*) echo "" ;;
esac
}
# ── GPUs ──
# lspci -nnD outputs:
# 0000:01:00.0 VGA compatible controller [0300]: NVIDIA Corporation GP107GL [Quadro P620] [10de:1cb6] (rev a1)
# We pick anything classified as VGA/3D/Display (display controllers).
gpu_array='[]'
while IFS= read -r line; do
[[ -z "$line" ]] && continue
pci_address="$(printf '%s' "$line" | awk '{print $1}')"
pci_id="$(printf '%s' "$line" | grep -oE '\[[0-9a-f]{4}:[0-9a-f]{4}\]' | tail -1 | tr -d '[]')"
# Description: everything between the "controller]:" header and the
# final "[pci_id]" tag. For AMD this includes the [AMD/ATI] tag; for
# NVIDIA/Intel it's just vendor + model.
desc="$(printf '%s' "$line" | sed -nE "s@.*\]:[[:space:]]*(.*)[[:space:]]+\[[0-9a-f]{4}:[0-9a-f]{4}\].*@\1@p")"
# Vendor classification
case "$desc" in
*NVIDIA*) vendor="NVIDIA" ;;
*"Advanced Micro Devices"*|*AMD*) vendor="AMD" ;;
*"Intel Corporation"*|*Intel*) vendor="Intel" ;;
*) vendor="Other" ;;
esac
# Model: strip every known vendor prefix from desc. Order matters —
# the longest specific prefix (AMD's "Inc. [AMD/ATI]") must come before
# the generic short one.
model="$(printf '%s' "$desc" | sed -E '
s/^Advanced Micro Devices, Inc\. \[AMD\/ATI\][[:space:]]+//
s/^Advanced Micro Devices(, Inc\.)?[[:space:]]+//
s/^NVIDIA Corporation[[:space:]]+//
s/^Intel Corporation[[:space:]]+//
s/[[:space:]]+$//
')"
# Kernel driver in use (may be empty if module not loaded yet)
kernel_driver="$(lspci -nnks "$pci_address" 2>/dev/null | awk -F: '/Kernel driver in use/{sub(/^[ \t]+/,"",$2); print $2; exit}')"
# Passthrough eligible if the GPU is bound to vfio-pci OR it's a discrete
# secondary GPU (not the primary console). Pragmatic heuristic: discrete
# GPUs are usually eligible; iGPUs (Intel HD/UHD, AMD APU iGPUs) usually not
# because they drive the host console.
passthrough_eligible=false
case "$kernel_driver" in
vfio-pci) passthrough_eligible=true ;;
nvidia|nouveau) passthrough_eligible=true ;; # discrete by definition
esac
# ProxMenux installer for this GPU vendor
proxmenux_installer="$(gpu_installer_for "$vendor")"
# Installed driver version from the managed_installs registry
installed_driver_version=""
if [[ "$vendor" == "NVIDIA" ]] && [[ -f /usr/local/share/proxmenux/managed_installs.json ]]; then
installed_driver_version="$(jq -r '
.items[]
| select(.removed_at == null and .type == "nvidia_xfree86")
| .current_version // ""
' /usr/local/share/proxmenux/managed_installs.json 2>/dev/null | head -1)"
fi
gpu_array="$(jq --argjson acc "$gpu_array" \
--arg vendor "$vendor" \
--arg model "$model" \
--arg pci_address "$pci_address" \
--arg pci_id "$pci_id" \
--arg kernel_driver "$kernel_driver" \
--argjson passthrough_eligible "$passthrough_eligible" \
--arg proxmenux_installer "$proxmenux_installer" \
--arg installed_driver_version "$installed_driver_version" \
-n '
$acc + [{
vendor: $vendor,
model: $model,
pci_address: $pci_address,
pci_id: $pci_id,
kernel_driver: (if $kernel_driver == "" then null else $kernel_driver end),
passthrough_eligible: $passthrough_eligible,
proxmenux_installer: (if $proxmenux_installer == "" then null else $proxmenux_installer end),
installed_driver_version: (if $installed_driver_version == "" then null else $installed_driver_version end)
}]
')"
done < <(lspci -nnD 2>/dev/null | grep -E 'VGA compatible|3D controller|Display controller' || true)
# ── TPUs (Google Coral) ──
# PCIe variant: vendor 1ac1 (Global Unichip Corp) is the Coral M.2 / mPCIe.
# USB variant: vendor 18d1 product 9302 (Google).
tpu_array='[]'
# PCIe Coral
while IFS= read -r line; do
[[ -z "$line" ]] && continue
pci_address="$(printf '%s' "$line" | awk '{print $1}')"
pci_id="$(printf '%s' "$line" | grep -oE '\[[0-9a-f]{4}:[0-9a-f]{4}\]' | tail -1 | tr -d '[]')"
tpu_array="$(jq --argjson acc "$tpu_array" \
--arg model "Coral PCIe" \
--arg pci_address "$pci_address" \
-n '
$acc + [{
vendor: "Google",
model: $model,
bus: "PCIe",
pci_address: $pci_address,
proxmenux_installer: "scripts/gpu_tpu/install_coral.sh",
installed_version: null
}]
')"
done < <(lspci -nnD 2>/dev/null | grep -iE '1ac1:|global unichip' || true)
# USB Coral
if command -v lsusb >/dev/null 2>&1; then
if lsusb 2>/dev/null | grep -qE '18d1:9302|Google.*Coral'; then
tpu_array="$(jq --argjson acc "$tpu_array" \
-n '
$acc + [{
vendor: "Google",
model: "Coral USB",
bus: "USB",
pci_address: null,
proxmenux_installer: "scripts/gpu_tpu/install_coral.sh",
installed_version: null
}]
')"
fi
fi
# ── NICs ──
# We want PHYSICAL interfaces (skip lo, veth*, tap*, fwln*, fwbr*, fwpr*).
# Also distinguish wired from wireless.
nic_array='[]'
wireless_array='[]'
# Map each interface → its bridge by walking /sys/class/net/<bridge>/brif/.
# We use bash glob expansion instead of `find -path` because find doesn't
# follow the symlinks under /sys cleanly.
declare -A bridge_for
for brif_dir in /sys/class/net/*/brif; do
[[ -d "$brif_dir" ]] || continue
bridge="$(basename "$(dirname "$brif_dir")")"
for member_link in "$brif_dir"/*; do
[[ -e "$member_link" ]] || continue
member="$(basename "$member_link")"
bridge_for["$member"]="$bridge"
done
done
# Iterate over each physical net device
for dev_path in /sys/class/net/*; do
ifname="$(basename "$dev_path")"
case "$ifname" in
lo|veth*|tap*|fwln*|fwbr*|fwpr*|vmbr*|bond*) continue ;;
esac
# Bridges and bonds we record as their own thing; PHY interfaces only here.
# Detect virtual interfaces (no device symlink → virtual)
[[ ! -e "$dev_path/device" ]] && continue
mac="$(cat "$dev_path/address" 2>/dev/null || echo "")"
[[ -z "$mac" ]] && continue
operstate="$(cat "$dev_path/operstate" 2>/dev/null | tr '[:lower:]' '[:upper:]' || echo "UNKNOWN")"
case "$operstate" in
UP|DOWN) ;;
*) operstate="UNKNOWN" ;;
esac
kernel_driver="$(basename "$(readlink "$dev_path/device/driver" 2>/dev/null || echo "")")"
# Wireless detection
if [[ -d "$dev_path/wireless" ]] || [[ -d "$dev_path/phy80211" ]]; then
wireless_array="$(jq --argjson acc "$wireless_array" \
--arg ifname "$ifname" \
--arg mac "$mac" \
-n '$acc + [{ifname: $ifname, mac: $mac}]')"
continue
fi
# Bridge membership: which vmbr* contains this NIC?
in_bridges_json='[]'
if [[ -n "${bridge_for[$ifname]:-}" ]]; then
in_bridges_json="$(jq -n --arg b "${bridge_for[$ifname]}" '[$b]')"
fi
nic_array="$(jq --argjson acc "$nic_array" \
--arg ifname "$ifname" \
--arg mac "$mac" \
--arg kernel_driver "$kernel_driver" \
--argjson in_bridges "$in_bridges_json" \
--arg operstate "$operstate" \
-n '
$acc + [{
ifname: $ifname,
mac: $mac,
kernel_driver: (if $kernel_driver == "" then null else $kernel_driver end),
in_bridges: $in_bridges,
operstate: $operstate
}]
')"
done
# Compose the final object
jq -n \
--argjson gpu "$gpu_array" \
--argjson tpu "$tpu_array" \
--argjson nic "$nic_array" \
--argjson wireless "$wireless_array" \
'{ gpu: $gpu, tpu: $tpu, nic: $nic, wireless: $wireless }'
+67
View File
@@ -0,0 +1,67 @@
#!/usr/bin/env bash
# ==========================================================
# ProxMenux backup manifest collector — kernel_params
# ==========================================================
# /proc/cmdline (filtered to user-meaningful extras), /etc/modules,
# and /etc/modprobe.d/ files with custom directives. Read-only.
# Schema: scripts/backup_restore/schema/manifest.schema.json
# ==========================================================
set -euo pipefail
# ── cmdline_extra ──
# /proc/cmdline contains the kernel command line the bootloader passed.
# We strip the boring boilerplate (BOOT_IMAGE, initrd, root, ro, rw, quiet,
# splash, boot=zfs, rootflags) so the manifest captures only the user-
# meaningful tweaks (intel_iommu, iommu=pt, hugepages, pcie_acs_override,
# acpi=off, etc.). These are the bits a restore wizard cares about.
cmdline_extra='[]'
if [[ -r /proc/cmdline ]]; then
raw_cmdline="$(cat /proc/cmdline)"
for token in $raw_cmdline; do
case "$token" in
BOOT_IMAGE=*|initrd=*|root=*|ro|rw|quiet|splash|boot=*|rootflags=*)
;; # boilerplate, drop
*)
cmdline_extra="$(jq --argjson acc "$cmdline_extra" --arg t "$token" -n '$acc + [$t]')"
;;
esac
done
fi
# ── modules_loaded_at_boot ──
# /etc/modules lists modules systemd-modules-load.service inserts on boot.
modules_at_boot='[]'
if [[ -r /etc/modules ]]; then
while IFS= read -r mod; do
# Strip comments and inline comments
mod="${mod%%#*}"
mod="$(printf '%s' "$mod" | xargs)"
[[ -z "$mod" ]] && continue
modules_at_boot="$(jq --argjson acc "$modules_at_boot" --arg m "$mod" -n '$acc + [$m]')"
done < /etc/modules
fi
# ── modprobe_d_files ──
# /etc/modprobe.d/*.conf files. We emit the path of every file that
# contains at least one `options`, `blacklist`, `install`, `alias`, or
# `softdep` directive — i.e. anything that has actual effect. Files that
# are empty or pure comments aren't worth tracking.
modprobe_files='[]'
if [[ -d /etc/modprobe.d ]]; then
for f in /etc/modprobe.d/*.conf; do
[[ -r "$f" ]] || continue
if grep -qE '^[[:space:]]*(options|blacklist|install|alias|softdep)[[:space:]]' "$f" 2>/dev/null; then
modprobe_files="$(jq --argjson acc "$modprobe_files" --arg p "$f" -n '$acc + [$p]')"
fi
done
fi
jq -n \
--argjson cmdline_extra "$cmdline_extra" \
--argjson modules_loaded "$modules_at_boot" \
--argjson modprobe_files "$modprobe_files" \
'{
cmdline_extra: $cmdline_extra,
modules_loaded_at_boot: $modules_loaded,
modprobe_d_files: $modprobe_files
}'
@@ -0,0 +1,81 @@
#!/usr/bin/env bash
# ==========================================================
# ProxMenux backup manifest collector — proxmenux_installed_components
# ==========================================================
# Reads ProxMenux's managed_installs registry + post-install
# tools marker file and emits the installed components array.
# Read-only. Schema:
# scripts/backup_restore/schema/manifest.schema.json
# ==========================================================
set -euo pipefail
REGISTRY="/usr/local/share/proxmenux/managed_installs.json"
INSTALLED_TOOLS="/usr/local/share/proxmenux/installed_tools.json"
components='[]'
# ── managed_installs registry ──
# Each entry already carries the installer path under `menu_script`,
# so we trust the registry as the single source of truth. We skip LXC
# entries because containers are restored via vzdump, not via the
# host-config restore path.
if [[ -r "$REGISTRY" ]]; then
while IFS= read -r item; do
[[ -z "$item" ]] && continue
id="$(printf '%s' "$item" | jq -r '.id')"
type="$(printf '%s' "$item" | jq -r '.type // ""')"
version="$(printf '%s' "$item" | jq -r '.current_version // ""')"
# menu_script in the registry is null for components that handle their
# own update lifecycle (e.g. OCI apps via the secure-gateway runtime).
# We keep that null forward: restore won't try to reinstall those —
# the user reconfigures them after restore.
installer="$(printf '%s' "$item" | jq -r '.menu_script // ""')"
components="$(jq --argjson acc "$components" \
--arg id "$id" --arg type "$type" --arg version "$version" --arg installer "$installer" \
-n '
$acc + [{
id: $id,
type: $type,
version_at_backup: (if $version == "" then null else $version end),
proxmenux_installer: (if $installer == "" then null else $installer end),
applied_settings: []
}]
')"
done < <(jq -c '.items[]? | select(.removed_at == null) | select(.type != "lxc")' "$REGISTRY" 2>/dev/null || true)
fi
# ── installed_tools.json (post-install optimizations) ──
# Format: array of {name: ..., installed_at: ...} or similar. The exact
# shape varies across ProxMenux versions; we emit one synthetic component
# named "post_install_optimizations" with the applied_settings list.
if [[ -r "$INSTALLED_TOOLS" ]]; then
applied_settings="$(jq -c '
if type == "object" then
(.tools // .installed // [] | map(.name // .id // tostring))
elif type == "array" then
map(.name // .id // tostring)
else []
end
' "$INSTALLED_TOOLS" 2>/dev/null || echo '[]')"
# Only emit if we have at least one applied setting — otherwise the
# component would be noise.
count="$(printf '%s' "$applied_settings" | jq 'length' 2>/dev/null || echo 0)"
if [[ "${count:-0}" -gt 0 ]]; then
components="$(jq --argjson acc "$components" --argjson s "$applied_settings" \
-n '
$acc + [{
id: "post_install_optimizations",
type: "proxmenux_post_install",
version_at_backup: null,
proxmenux_installer: "scripts/post_install/customizable_post_install.sh",
applied_settings: $s
}]
')"
fi
fi
# Output: bare array (not wrapped in an object — the orchestrator places
# this under .proxmenux_installed_components).
printf '%s\n' "$components"
+90
View File
@@ -0,0 +1,90 @@
#!/usr/bin/env bash
# ==========================================================
# ProxMenux backup manifest collector — source_host
# ==========================================================
# Emits the `source_host` section of the manifest as JSON to
# stdout. Read-only; no side effects. Schema:
# scripts/backup_restore/schema/manifest.schema.json
# ==========================================================
set -euo pipefail
# ── pve_version_full / pve_version ──
# pveversion's first line is like:
# pve-manager/9.2.2/b9984c6d90a4bd80 (running kernel: 7.0.2-6-pve)
pve_version_full=""
pve_version=""
if command -v pveversion >/dev/null 2>&1; then
pve_version_full="$(pveversion 2>/dev/null | head -1 || true)"
# Extract the X.Y.Z between "pve-manager/" and "/"
pve_version="$(printf '%s\n' "$pve_version_full" | sed -nE 's@^pve-manager/([0-9.]+)/.*@\1@p')"
fi
# ── pbs_version ──
# PBS is a separate package. If proxmox-backup-manager exists, host has PBS role.
pbs_version=""
if command -v proxmox-backup-manager >/dev/null 2>&1; then
pbs_version="$(proxmox-backup-manager versions 2>/dev/null | awk '/^proxmox-backup-server/{print $2; exit}' || true)"
fi
# ── roles ──
roles_json='[]'
if [[ -n "$pve_version" && -n "$pbs_version" ]]; then
roles_json='["pve","pbs"]'
elif [[ -n "$pve_version" ]]; then
roles_json='["pve"]'
elif [[ -n "$pbs_version" ]]; then
roles_json='["pbs"]'
else
# No PVE, no PBS — exit with the unknown sentinel. Caller decides
# whether to abort or generate a system-only manifest.
roles_json='[]'
fi
# ── kernel, boot_mode, root_fs ──
kernel="$(uname -r)"
if [[ -d /sys/firmware/efi ]]; then
boot_mode="efi"
else
boot_mode="bios"
fi
root_fs="$(findmnt -no FSTYPE / 2>/dev/null || echo ext4)"
# ── CPU model / arch ──
cpu_model="$(lscpu 2>/dev/null | awk -F: '/^Model name/{sub(/^[ \t]+/, "", $2); print $2; exit}')"
cpu_arch="$(uname -m)"
# Normalize to schema enum
case "$cpu_arch" in
x86_64|amd64) cpu_arch="x86_64" ;;
aarch64|arm64) cpu_arch="aarch64" ;;
esac
# ── memory_kb ──
memory_kb="$(awk '/^MemTotal:/{print $2; exit}' /proc/meminfo 2>/dev/null || echo 0)"
# Build JSON. Use --arg for strings (always quoted), --argjson for
# numbers/arrays/null. Empty strings → null per schema convention.
jq -n \
--arg hostname "$(hostname)" \
--arg pve_version "$pve_version" \
--arg pve_version_full "$pve_version_full" \
--arg pbs_version "$pbs_version" \
--argjson roles "$roles_json" \
--arg kernel "$kernel" \
--arg boot_mode "$boot_mode" \
--arg root_fs "$root_fs" \
--arg cpu_model "$cpu_model" \
--arg cpu_arch "$cpu_arch" \
--argjson memory_kb "$memory_kb" \
'{
hostname: $hostname,
pve_version: (if $pve_version == "" then null else $pve_version end),
pve_version_full: (if $pve_version_full == "" then null else $pve_version_full end),
pbs_version: (if $pbs_version == "" then null else $pbs_version end),
roles: $roles,
kernel: $kernel,
boot_mode: $boot_mode,
root_fs: $root_fs,
cpu_model: $cpu_model,
cpu_arch: $cpu_arch,
memory_kb: $memory_kb
}'
+252
View File
@@ -0,0 +1,252 @@
#!/usr/bin/env bash
# ==========================================================
# ProxMenux backup manifest collector — storage_inventory
# ==========================================================
# ZFS pools (with stable by-id devices), LVM VGs + thin pools,
# physical disks, PVE storage.cfg, and external mounts.
# Read-only. Schema:
# scripts/backup_restore/schema/manifest.schema.json
# ==========================================================
set -euo pipefail
# ── ZFS pools ──
zfs_pools='[]'
if command -v zpool >/dev/null 2>&1; then
while IFS= read -r pool; do
[[ -z "$pool" ]] && continue
# type: parse zpool status — first vdev line after 'config:' header.
# Single-device pool shows the device directly; mirror/raidz prefix the
# vdev type. We look at the indented children list.
pool_type="single"
devices='[]'
# `zpool status -P` outputs full /dev/disk/by-id/... paths for the
# member disks. We isolate the first whitespace-delimited token on
# each child line and decide:
# - vdev type lines (mirror-0, raidz1-0, stripe, ...) → pool type
# - leaf device lines (/dev/disk/by-id/* or /dev/sd*) → membership
while IFS= read -r vdev_line; do
token="$(printf '%s' "$vdev_line" | awk '{print $1}')"
[[ -z "$token" || "$token" == "NAME" || "$token" == "$pool" ]] && continue
case "$token" in
mirror-*) pool_type="mirror" ;;
raidz1-*) pool_type="raidz1" ;;
raidz2-*) pool_type="raidz2" ;;
raidz3-*) pool_type="raidz3" ;;
stripe-*) pool_type="stripe" ;;
/dev/disk/by-id/*)
# Strip the /dev/disk/by-id/ prefix for the schema field;
# leave any -partN suffix in place — the restore wizard uses
# the exact same string to look the disk back up.
dev_name="${token#/dev/disk/by-id/}"
devices="$(jq --argjson acc "$devices" --arg d "$dev_name" -n '$acc + [$d]')"
;;
/dev/*)
# Fallback: ZFS pool created with raw /dev/sdX paths. Record
# them as-is; restore will need to remap manually.
devices="$(jq --argjson acc "$devices" --arg d "$token" -n '$acc + [$d]')"
;;
esac
done < <(zpool status -P "$pool" 2>/dev/null | awk '/^config:/{flag=1; next} /^errors:/{flag=0} flag')
size_bytes="$(zpool list -H -p -o size "$pool" 2>/dev/null || echo 0)"
health="$(zpool list -H -o health "$pool" 2>/dev/null || echo UNKNOWN)"
compression="$(zfs get -H -o value compression "$pool" 2>/dev/null || echo "")"
mountpoint="$(zfs get -H -o value mountpoint "$pool" 2>/dev/null || echo "")"
zfs_pools="$(jq --argjson acc "$zfs_pools" \
--arg name "$pool" \
--arg type "$pool_type" \
--argjson devices "$devices" \
--arg mountpoint "$mountpoint" \
--arg compression "$compression" \
--argjson size_bytes "${size_bytes:-0}" \
--arg health "$health" \
-n '
$acc + [{
name: $name,
type: $type,
devices_by_id: $devices,
mountpoint: $mountpoint,
compression: $compression,
size_bytes: $size_bytes,
health: $health
}]
')"
done < <(zpool list -H -o name 2>/dev/null || true)
fi
# ── LVM VGs + thin pools ──
lvm_vgs='[]'
if command -v vgs >/dev/null 2>&1; then
# vgs --reportformat json --units b is reliable in lvm2 ≥ 2.02
vg_json="$(vgs --reportformat json --units b --noheadings -o vg_name,vg_size 2>/dev/null || echo '{}')"
while IFS= read -r vg_name; do
[[ -z "$vg_name" || "$vg_name" == "null" ]] && continue
vg_size="$(printf '%s' "$vg_json" | jq -r --arg n "$vg_name" '.report[0].vg[]? | select(.vg_name == $n) | .vg_size' | sed 's/[Bb]$//' | head -1)"
# Thin pools in this VG
thin_pools='[]'
while IFS= read -r lv_line; do
[[ -z "$lv_line" ]] && continue
lv_name="$(printf '%s' "$lv_line" | awk '{print $1}')"
lv_size="$(printf '%s' "$lv_line" | awk '{print $2}' | sed 's/[Bb]$//')"
thin_pools="$(jq --argjson acc "$thin_pools" \
--arg n "$lv_name" --argjson s "${lv_size:-0}" \
-n '$acc + [{lv_name: $n, size_bytes: $s}]')"
done < <(lvs --noheadings --units b -o lv_name,lv_size --select "vg_name=$vg_name && lv_attr=~^t" 2>/dev/null || true)
lvm_vgs="$(jq --argjson acc "$lvm_vgs" \
--arg n "$vg_name" --argjson s "${vg_size:-0}" --argjson tp "$thin_pools" \
-n '$acc + [{name: $n, size_bytes: $s, thin_pools: $tp}]')"
done < <(printf '%s' "$vg_json" | jq -r '.report[0].vg[]?.vg_name' 2>/dev/null || true)
fi
# ── Physical disks (by-id resolution) ──
physical_disks='[]'
# Build name → by-id map by walking /dev/disk/by-id/. A single block
# device usually has multiple by-id symlinks (ata-*, wwn-*, scsi-*, …).
# We prefer the most human-readable identifier in this order:
# ata-* → nvme-* → scsi-* → usb-* → wwn-*
# This also makes the manifest consistent with what `zpool status -P`
# reports (zpool defaults to ata-* / wwn-* depending on bus).
declare -A by_id_for
declare -A by_id_priority_for
priority_for_id() {
case "$1" in
ata-*) echo 1 ;;
nvme-*) echo 2 ;;
scsi-*) echo 3 ;;
usb-*) echo 4 ;;
wwn-*) echo 5 ;;
*) echo 9 ;;
esac
}
if [[ -d /dev/disk/by-id ]]; then
for link in /dev/disk/by-id/*; do
[[ -L "$link" ]] || continue
by_id="$(basename "$link")"
# Skip partition symlinks — we want whole-disk only.
[[ "$by_id" == *-part* ]] && continue
target="$(basename "$(readlink -f "$link")")"
[[ -z "$target" ]] && continue
new_prio="$(priority_for_id "$by_id")"
cur_prio="${by_id_priority_for[$target]:-99}"
if (( new_prio < cur_prio )); then
by_id_for["$target"]="$by_id"
by_id_priority_for["$target"]="$new_prio"
fi
done
fi
# lsblk -d -b -J for whole disks
lsblk_json="$(lsblk -d -b -o NAME,MODEL,SIZE,TYPE -J 2>/dev/null || echo '{}')"
while IFS= read -r disk_line; do
[[ -z "$disk_line" ]] && continue
name="$(printf '%s' "$disk_line" | jq -r '.name')"
model="$(printf '%s' "$disk_line" | jq -r '.model // ""')"
size="$(printf '%s' "$disk_line" | jq -r '.size // 0')"
type="$(printf '%s' "$disk_line" | jq -r '.type')"
# Only PHYSICAL disks.
# - skip non-disk types (rom, loop)
# - skip zd* (ZFS zvols backing VMs)
# - skip dm-* (LVM-mapped devices)
# - skip loop* (defensive — type filter usually catches it)
[[ "$type" != "disk" ]] && continue
case "$name" in
zd*|dm-*|loop*) continue ;;
esac
by_id="${by_id_for[$name]:-}"
physical_disks="$(jq --argjson acc "$physical_disks" \
--arg n "$name" --arg m "$model" --argjson s "${size:-0}" --arg bid "$by_id" \
-n '
$acc + [{
name: $n,
model: (if $m == "" then null else $m end),
size_bytes: $s,
by_id: (if $bid == "" then null else $bid end)
}]
')"
done < <(printf '%s' "$lsblk_json" | jq -c '.blockdevices[]?' 2>/dev/null || true)
# ── PVE storage.cfg ──
# Format is whitespace-key-value with blank-line separators:
# <type>: <id>
# key value
# key value
pve_storage='[]'
if [[ -r /etc/pve/storage.cfg ]]; then
current_type=""; current_id=""; current_extra='{}'
flush() {
if [[ -n "$current_id" ]]; then
pve_storage="$(jq --argjson acc "$pve_storage" \
--arg id "$current_id" --arg t "$current_type" --argjson e "$current_extra" \
-n '$acc + [(($e) + {id: $id, type: $t})]')"
fi
current_type=""; current_id=""; current_extra='{}'
}
while IFS= read -r line; do
if [[ -z "${line// }" ]]; then
flush; continue
fi
if [[ "$line" =~ ^([a-z]+):[[:space:]]+([A-Za-z0-9_.-]+) ]]; then
flush
current_type="${BASH_REMATCH[1]}"
current_id="${BASH_REMATCH[2]}"
elif [[ "$line" =~ ^[[:space:]]+([a-z_]+)[[:space:]]+(.*)$ ]]; then
key="${BASH_REMATCH[1]}"
val="${BASH_REMATCH[2]}"
case "$key" in
# `content` is a comma-separated list — split into JSON array
content)
content_array="$(printf '%s\n' "$val" | tr ',' '\n' | jq -R . | jq -s .)"
current_extra="$(jq --argjson e "$current_extra" --argjson c "$content_array" -n '$e + {content: $c}')"
;;
*)
current_extra="$(jq --argjson e "$current_extra" --arg k "$key" --arg v "$val" -n '$e + {($k): $v}')"
;;
esac
fi
done < /etc/pve/storage.cfg
flush
fi
# ── External mounts (NFS/CIFS/etc.) ──
# Filter on filesystem types we care about for the manifest. Drop FUSE
# pmxcfs (/etc/pve), tmpfs, devtmpfs, autofs, ZFS internals already
# accounted for. NFS, CIFS, ISO mount points are the interesting ones.
mounts='[]'
if command -v findmnt >/dev/null 2>&1; then
while IFS= read -r mline; do
[[ -z "$mline" ]] && continue
target="$(printf '%s' "$mline" | jq -r '.target')"
source="$(printf '%s' "$mline" | jq -r '.source')"
fstype="$(printf '%s' "$mline" | jq -r '.fstype')"
options="$(printf '%s' "$mline" | jq -r '.options // ""')"
mounts="$(jq --argjson acc "$mounts" \
--arg t "$target" --arg s "$source" --arg f "$fstype" --arg o "$options" \
-n '
$acc + [{
target: $t,
source: $s,
fstype: $f,
options: (if $o == "" then null else $o end)
}]
')"
done < <(findmnt -t nfs,nfs4,cifs,smbfs,fuseblk,fuse.glusterfs -J 2>/dev/null \
| jq -c '.. | objects | select(.target?)' 2>/dev/null \
| grep -vE '"target":"/etc/pve"' || true)
fi
# Compose
jq -n \
--argjson zfs_pools "$zfs_pools" \
--argjson lvm_vgs "$lvm_vgs" \
--argjson physical_disks "$physical_disks" \
--argjson pve_storage "$pve_storage" \
--argjson mounts "$mounts" \
'{
zfs_pools: $zfs_pools,
lvm: { vgs: $lvm_vgs },
physical_disks: $physical_disks,
pve_storage_cfg: $pve_storage,
mounts: $mounts
}'

Some files were not shown because too many files have changed in this diff Show More