From b45799d31974431657cbe6c60bdd1c7b1b859318 Mon Sep 17 00:00:00 2001 From: MacRimi Date: Sat, 15 Aug 2026 22:11:41 +0200 Subject: [PATCH] GPU realtime: cache endpoint TTL 4s + trim intel_gpu_top warmup --- AppImage/components/virtual-machines.tsx | 58 +++++++++++++--- AppImage/messages/es/common.json | 4 +- AppImage/scripts/flask_server.py | 85 ++++++++++++------------ 3 files changed, 91 insertions(+), 56 deletions(-) diff --git a/AppImage/components/virtual-machines.tsx b/AppImage/components/virtual-machines.tsx index d0a76ffd..dea55583 100644 --- a/AppImage/components/virtual-machines.tsx +++ b/AppImage/components/virtual-machines.tsx @@ -107,6 +107,11 @@ interface LxcAppWatch { // Updates tab helper-scripts section find its matching registered // app to pull installed/upstream version data from. helper_slug?: string + // Per-app opt-out of the CT's aggregate updates badge (default: + // included). Independent from the app's notification toggle. + exclude_from_badge?: boolean + // Per-app opt-out for the `app_update_available` notification. + notifications_enabled?: boolean } interface VMData { @@ -1960,6 +1965,31 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => { // (mirrors the Secure Gateway visual treatment). Security count // stays red because it's still an urgency cue independent of the // update theme. + // Aggregate updates counter (OS packages + registered apps that + // aren't opted out of the badge). Returns an update_check-shaped + // object so `renderLxcUpdateBadge` and the tab count can consume it + // without knowing about app entries. The underlying `update_check` + // stays strictly OS-only on the backend — the Updates tab's "OS + // packages" section needs a clean `available`/`count` to avoid a + // false "N package pending" every time a registered app has an + // upstream bump. + const getAggregateUpdateCheck = (vm: VMData): LxcUpdateCheck | undefined => { + const uc = vm.update_check + const appCount = (vm.app_watches || []).filter( + (a) => a.update_available === true && !a.exclude_from_badge, + ).length + const osCount = uc?.count ?? 0 + const total = osCount + appCount + if (!uc && appCount === 0) return undefined + if (total === 0) return uc + return { + ...(uc || {}), + count: total, + available: true, + last_check: uc?.last_check ?? new Date().toISOString(), + } as LxcUpdateCheck + } + const renderLxcUpdateBadge = ( uc?: LxcUpdateCheck, compact = false, @@ -2587,7 +2617,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => { })} {vm.type === "lxc" && (
- {renderLxcUpdateBadge(vm.update_check)} + {renderLxcUpdateBadge(getAggregateUpdateCheck(vm))}
)} @@ -2751,9 +2781,12 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => { /> ) })} - {vm.type === "lxc" && vm.update_check?.available && (vm.update_check?.count ?? 0) > 0 && ( - - )} + {vm.type === "lxc" && (() => { + const agg = getAggregateUpdateCheck(vm) + return agg?.available && (agg.count ?? 0) > 0 ? ( + + ) : null + })()} @@ -2884,7 +2917,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => { Uptime / Type / Status chips. */} {selectedVM.type === "lxc" && renderLxcUpdateBadge( - selectedVM.update_check, + getAggregateUpdateCheck(selectedVM), false, () => setActiveModalTab("updates"), )} @@ -2931,7 +2964,7 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => { )} {selectedVM.type === "lxc" && renderLxcUpdateBadge( - selectedVM.update_check, + getAggregateUpdateCheck(selectedVM), false, () => setActiveModalTab("updates"), )} @@ -3011,11 +3044,14 @@ const handleDownloadLogs = async (vmid: number, vmName: string) => { {t("vmLxc.tabs.updates")} - {typeof selectedVM.update_check?.count === "number" && selectedVM.update_check.count > 0 && ( - - {selectedVM.update_check.count} - - )} + {(() => { + const agg = getAggregateUpdateCheck(selectedVM) + return typeof agg?.count === "number" && agg.count > 0 ? ( + + {agg.count} + + ) : null + })()} )} {/* Mount Points tab — LXC only, and only when at least diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index eb423b2a..e12d4d47 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -3315,8 +3315,8 @@ "switchMode": "Modo de cambio", "currentDriver": "Controlador actual", "consumer": "Consumidor", - "graphicsClock": "Reloj de gráficos", - "memoryClock": "Reloj de memoria", + "graphicsClock": "Frecuencia GPU", + "memoryClock": "Frecuencia de memoria", "powerDraw": "Consumo de energía", "temperature": "Temperatura", "memory": "Memoria", diff --git a/AppImage/scripts/flask_server.py b/AppImage/scripts/flask_server.py index 39bddc62..51cc606c 100644 --- a/AppImage/scripts/flask_server.py +++ b/AppImage/scripts/flask_server.py @@ -6215,36 +6215,20 @@ def get_proxmox_vms(): # apps per CT (0..N). Populates header badge, # Updates modal connected row, and the App # tab. Absent key = no apps registered. + # The list-card aggregate badge folds app + # updates on the frontend side so the + # OS-only `update_check.available` / `.count` + # stay clean — the Updates tab reads them to + # decide whether "OS packages pending" + + # "Apply OS update" should show, and + # inflating them with app pending was + # producing a false "1 package pending" + # every time a registered app had a newer + # upstream version. app_list = lxc_app_map.get(str(resource.get('vmid'))) if app_list: vm_data['app_watches'] = app_list - # Fold registered-app updates into the CT's - # aggregate updates badge so the list card - # counter reflects OS + apps in one number. - # Apps flagged `exclude_from_badge` are - # omitted from the count (pinned versions, - # tracker-locked apps, etc.) — see the - # validator in lxc_apps.py for the full - # rationale. Independent from - # `notifications_enabled`. - if app_list: - app_upd_count = sum( - 1 for a in app_list - if a.get('update_available') is True - and not a.get('exclude_from_badge') - ) - if app_upd_count: - uc = vm_data.get('update_check') or {} - # Synthesize a minimal update_check - # entry when the CT has no apt/apk - # data (OCI, non-Debian, checker off) - # but at least one counted app. - uc = dict(uc) if uc else {} - uc['count'] = int(uc.get('count') or 0) + app_upd_count - uc['available'] = True - vm_data['update_check'] = uc - # PVE's cluster resources API reports disk=0 for most # QEMU VMs — it can't see inside the guest filesystem # for the common storage backends. For running QEMU @@ -7015,12 +6999,16 @@ def get_detailed_gpu_info(gpu): # print(f"[v0] Process started with PID: {process.pid}", flush=True) pass - # print(f"[v0] Waiting 1 second for intel_gpu_top to initialize and detect processes...", flush=True) - pass - time.sleep(1) - + # intel_gpu_top needs a small warmup for the first JSON + # object to hit stdout. 300 ms is enough on every host + # tested — the previous 1 s was tuned when the tool was + # slower to boot and doubled the modal open latency for + # no gain. Combined with the shorter read timeout below + # this halves the worst-case blocking time. + time.sleep(0.3) + start_time = time.time() - timeout = 3 + timeout = 1.5 json_objects = [] buffer = "" brace_count = 0 @@ -13773,30 +13761,40 @@ def api_hardware_live(): return jsonify({'error': str(e)}), 500 +_gpu_realtime_cache: dict[str, tuple[float, dict]] = {} +_gpu_realtime_cache_lock = threading.Lock() +# Frontend polls this endpoint every 3 s per open GPU modal. For +# Intel and AMD the underlying tool call (intel_gpu_top / rocm-smi) +# blocks ~2 s per invocation, so a naked request-per-poll makes the +# modal feel sluggish and stacks CPU. A 4 s TTL means the second and +# third poll of any 4 s window serve straight from memory while the +# first still pays the tool cost; NVIDIA (nvidia-smi ~200 ms) also +# benefits by dropping the second nvidia-smi spawn. +_GPU_REALTIME_TTL = 4.0 + @app.route('/api/gpu//realtime', methods=['GET']) @require_auth def api_gpu_realtime(slot): """Get real-time GPU monitoring data for a specific GPU""" try: - # print(f"[v0] /api/gpu/{slot}/realtime - Getting GPU info...") - pass - + now = time.time() + with _gpu_realtime_cache_lock: + hit = _gpu_realtime_cache.get(slot) + if hit and (now - hit[0]) < _GPU_REALTIME_TTL: + return jsonify(hit[1]) + gpus = get_gpu_info() - + gpu = None for g in gpus: # Match by slot or if the slot is a substring of the GPU's slot (e.g., '00:01.0' matching '00:01') if g.get('slot') == slot or slot in g.get('slot', ''): gpu = g break - + if not gpu: - # print(f"[v0] GPU with slot matching '{slot}' not found") - pass return jsonify({'error': 'GPU not found'}), 404 - - # print(f"[v0] Getting detailed monitoring data for GPU at slot {gpu.get('slot')}...") - pass + detailed_info = get_detailed_gpu_info(gpu) gpu.update(detailed_info) @@ -13840,10 +13838,11 @@ def api_gpu_realtime(slot): 'sriov_consumer': gpu.get('sriov_consumer'), } + with _gpu_realtime_cache_lock: + _gpu_realtime_cache[slot] = (time.time(), realtime_data) + return jsonify(realtime_data) except Exception as e: - # print(f"[v0] Error getting real-time GPU data: {e}") - pass import traceback traceback.print_exc() return jsonify({'error': str(e)}), 500