feat(oci): run official container images as native LXC containers

Adds the OCI manager: an engine that turns a Docker Compose file into an
LXC definition, a catalog of 365 applications drawn from LinuxServer.io
and other container image sources, and a per-instance registry recording
what each container was built from. Reachable from the main menu.

Catalog text is translated like every other string in the project: the
taglines go through translate() and land in lang/*.json, so the entries
read in all eight languages instead of only English.

Translation cache builder:
- a failed translation leaves the key absent rather than writing English,
  which previously made the string count as translated forever
- a result identical to a 3+ word source is rejected, catching a provider
  that silently returns the text it was given
- strings that are nothing but glossary terms keep their source spelling
  instead of being discarded as failures
- no backoff between attempts when the provider is deterministic
- application names are protected so "HAOS One" survives translation
- argos joins the provider list, and the workflow reads the OCI sources

Audit & Report:
- findings that moved in the wrong direction between runs are reported
  alongside the ones that improved
- an accepted risk can carry a review date and is flagged when it falls due
- backup checks explain in plain language what they looked at and what to
  do next

Monitor:
- disks can be excluded from periodic reads, and an idle disk says so
  instead of showing a stale temperature
- per-disk identity survives a controller or enclosure change
- scheduled Borg backups resolve their SSH key from the repository entry
- PVE upgrades log the package list and the resulting dpkg changes

The web build no longer copies scripts/ into public/: the documentation
links to GitHub, so nothing read that folder.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
MacRimi
2026-09-22 18:24:59 +02:00
co-authored by Claude Opus 5
parent b36498f215
commit bcabcb618c
670 changed files with 221410 additions and 215 deletions
+49 -1
View File
@@ -691,6 +691,37 @@ def run_assessment(profile: str = "full",
return run_id
def _scope(finding: dict) -> int:
"""How many objects a finding covers. A check that named three guests
and now names nine describes a larger problem, even at the same
gravity."""
return len(finding.get("affected") or [])
def _movement(previous: dict, current: dict) -> int:
"""Whether a finding present in both runs got worse (1), better (-1) or
held (0). Gravity decides; scope only breaks a tie, because a finding
takes the gravity of its gravest object and dropping from critical to
warning is progress however many objects it now names."""
before = audit_store.CLASS_ORDER.get(previous["classification"])
after = audit_store.CLASS_ORDER.get(current["classification"])
if before is not None and after is not None and before != after:
return 1 if after < before else -1
before_scope, after_scope = _scope(previous), _scope(current)
if after_scope != before_scope:
return 1 if after_scope > before_scope else -1
return 0
def _against(current: dict, previous: dict) -> dict:
"""A finding carrying where it came from, so the reader is told what
moved instead of only what it is now."""
return {**current,
"previous_classification": previous["classification"],
"previous_affected": _scope(previous),
"affected_count": _scope(current)}
def compare_runs(base_run: str, other_run: str) -> dict[str, list[dict]]:
"""Classify how findings moved between two runs.
@@ -700,6 +731,12 @@ def compare_runs(base_run: str, other_run: str) -> dict[str, list[dict]]:
that merges them would tell its reader the problem went away when the
decision was to live with it.
A finding that was already failing and still fails is never new. It
either got worse, got better without being resolved, or held: reporting
a warning that became critical as new hides that it was already there,
and reporting a critical that dropped to a warning as new tells the
reader their work created a problem.
``unchanged`` is kept so a report can state that the rest of the
surface held steady rather than leaving it unaccounted for.
"""
@@ -708,6 +745,7 @@ def compare_runs(base_run: str, other_run: str) -> dict[str, list[dict]]:
other = {f["check_id"]: f for f in audit_store.get_findings(other_run)}
new, resolved, accepted, unchanged, unverified = [], [], [], [], []
worse, better = [], []
for check_id, current in other.items():
previous = base.get(check_id)
was = previous["classification"] in problems if previous else False
@@ -718,8 +756,16 @@ def compare_runs(base_run: str, other_run: str) -> dict[str, list[dict]]:
unverified.append(current)
elif now and current.get("decision") == audit_store.DECISION_ACCEPTED:
accepted.append(current)
elif now and (not was or previous["classification"] != current["classification"]):
elif now and not was:
new.append(current)
elif now and was:
moved = _movement(previous, current)
if moved > 0:
worse.append(_against(current, previous))
elif moved < 0:
better.append(_against(current, previous))
else:
unchanged.append(current)
elif was and not now:
if current.get("decision") == audit_store.DECISION_ACCEPTED:
accepted.append(current)
@@ -738,6 +784,8 @@ def compare_runs(base_run: str, other_run: str) -> dict[str, list[dict]]:
return {
"new": new,
"worse": worse,
"better": better,
"resolved": resolved,
"accepted": accepted,
"unchanged": unchanged,
+29 -7
View File
@@ -229,12 +229,19 @@ def init_db() -> None:
-- Accepted risks outlive the run that surfaced them, so they
-- are keyed by check rather than by finding. expires_at NULL
-- means the acceptance does not lapse on its own.
--
-- review_at is deliberately not expires_at. Expiry withdraws
-- the decision and the finding becomes a problem again on its
-- own; a review date leaves the decision standing and only
-- brings it back to the reader, so "remind me in a year" no
-- longer has to be spelled as "stop accepting this in a year".
CREATE TABLE IF NOT EXISTS audit_exceptions (
check_id TEXT PRIMARY KEY,
reason TEXT NOT NULL,
accepted_by TEXT NOT NULL,
accepted_at INTEGER NOT NULL,
expires_at INTEGER
expires_at INTEGER,
review_at INTEGER
);
CREATE INDEX IF NOT EXISTS idx_audit_findings_run
@@ -250,7 +257,7 @@ def init_db() -> None:
"audit_findings": {"raw_state": "TEXT", "exception_snapshot": "TEXT",
"scope": "TEXT", "details": "TEXT", "classification": "TEXT",
"raw_classification": "TEXT", "decision": "TEXT"},
"audit_exceptions": {"scope": "TEXT"},
"audit_exceptions": {"scope": "TEXT", "review_at": "INTEGER"},
}.items():
present = {row[1] for row in conn.execute(f"PRAGMA table_info({table})")}
for name, kind in columns.items():
@@ -503,12 +510,17 @@ def check_history(check_id: str, limit: int = 30) -> list[dict[str, Any]]:
# ---------------------------------------------------------------------------
def accept_risk(check_id: str, reason: str, accepted_by: str,
expires_at: Optional[int] = None, *, scope: str) -> None:
expires_at: Optional[int] = None, *, scope: str,
review_at: Optional[int] = None) -> None:
"""Record a deliberate decision to leave a finding unresolved.
A reason is mandatory: an acceptance without one is indistinguishable
from having silenced the check, which is what this register exists to
prevent.
``review_at`` asks to be reminded of the decision on a date without
withdrawing it. It is independent of ``expires_at``: an acceptance
can stand indefinitely and still come back for review.
"""
if not (reason or "").strip():
raise ValueError("an accepted risk requires a reason")
@@ -516,18 +528,21 @@ def accept_risk(check_id: str, reason: str, accepted_by: str,
raise ValueError("an accepted risk requires an assessed scope")
if expires_at is not None and expires_at <= time.time():
raise ValueError("expiry must be in the future")
if review_at is not None and review_at <= time.time():
raise ValueError("the review date must be in the future")
init_db()
conn = _connect()
try:
conn.execute("BEGIN IMMEDIATE")
decision = dict(check_id=check_id, reason=reason.strip(), accepted_by=accepted_by,
accepted_at=int(time.time()), expires_at=expires_at, scope=scope)
accepted_at=int(time.time()), expires_at=expires_at, scope=scope,
review_at=review_at)
conn.execute(
"INSERT OR REPLACE INTO audit_exceptions "
"(check_id, reason, accepted_by, accepted_at, expires_at, scope) "
"VALUES (?, ?, ?, ?, ?, ?)",
"(check_id, reason, accepted_by, accepted_at, expires_at, scope, review_at) "
"VALUES (?, ?, ?, ?, ?, ?, ?)",
(check_id, reason.strip(), accepted_by, int(time.time()),
expires_at, scope),
expires_at, scope, review_at),
)
conn.execute("INSERT INTO audit_exception_events (check_id, action, happened_at, decision) "
"VALUES (?, 'accepted', ?, ?)",
@@ -643,6 +658,13 @@ def all_exceptions() -> list[dict[str, Any]]:
item["lapsed"] = bool(
item["expires_at"] is not None and item["expires_at"] <= now
)
# Due for review, and still in force: the decision holds, it is
# only asking to be looked at again.
item["review_due"] = bool(
item.get("review_at") is not None
and item["review_at"] <= now
and not item["lapsed"]
)
out.append(item)
return out
finally:
+1
View File
@@ -138,6 +138,7 @@ cp "$SCRIPT_DIR/mount_monitor.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠
cp "$SCRIPT_DIR/lxc_mount_points.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ lxc_mount_points.py not found"
cp "$SCRIPT_DIR/disk_temperature_history.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ disk_temperature_history.py not found"
cp "$SCRIPT_DIR/smartctl_resolver.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ smartctl_resolver.py not found"
cp "$SCRIPT_DIR/disk_identity.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ disk_identity.py not found"
cp "$SCRIPT_DIR/health_thresholds.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ health_thresholds.py not found"
cp "$SCRIPT_DIR/managed_installs.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ managed_installs.py not found"
cp "$SCRIPT_DIR/lxc_apps.py" "$APP_DIR/usr/bin/" 2>/dev/null || echo "⚠️ lxc_apps.py not found"
+76
View File
@@ -0,0 +1,76 @@
"""
Physical disks and a stable identity for each, read without touching them.
Everything here comes from udev's database and /sys through lsblk, which
reads these columns without opening the block device. A drive that is
asleep, or idle and about to be, is not disturbed by being listed.
The identity matters because a kernel name is not one: a USB drive can be
sda on one boot and sdb on the next, so anything the user attaches to a
disk has to follow the disk, not the letter it happened to get.
"""
import re
import subprocess
from typing import Any, Dict, List
_LSBLK_TIMEOUT = 5
_FIELD_RE = re.compile(r'(\w+)="([^"]*)"')
_SKIP_PREFIXES = ("loop", "zd", "nbd", "ram", "sr")
def disk_key(serial: str, wwn: str, name: str) -> str:
"""Stable identity: the serial where udev knows one, then the WWN,
and only as a last resort the kernel name."""
serial = (serial or "").strip()
wwn = (wwn or "").strip()
if serial:
return f"serial:{serial}"
if wwn:
return f"wwn:{wwn}"
return f"name:{name}"
def list_physical_disks() -> List[Dict[str, Any]]:
"""Every physical disk with its identity and the facts the interface
shows about it. Returns an empty list if lsblk cannot be read."""
try:
proc = subprocess.run(
["lsblk", "-d", "-n", "-P", "-b", "-o",
"NAME,TYPE,MODEL,SERIAL,WWN,SIZE,TRAN,ROTA"],
capture_output=True, text=True, timeout=_LSBLK_TIMEOUT,
)
except (subprocess.TimeoutExpired, OSError):
return []
if proc.returncode != 0:
return []
disks: List[Dict[str, Any]] = []
for line in proc.stdout.splitlines():
fields = dict(_FIELD_RE.findall(line))
name = fields.get("NAME", "")
if fields.get("TYPE") != "disk" or not name or name.startswith(_SKIP_PREFIXES):
continue
serial = fields.get("SERIAL", "").strip()
wwn = fields.get("WWN", "").strip()
try:
size = int(fields.get("SIZE") or 0)
except ValueError:
size = 0
disks.append({
"name": name,
"key": disk_key(serial, wwn, name),
"model": fields.get("MODEL", "").strip(),
"serial": serial,
"size_bytes": size,
"transport": fields.get("TRAN", "").strip(),
"rotational": fields.get("ROTA", "").strip() == "1",
})
return disks
def names_for_keys(keys) -> set:
"""Current kernel names of the disks whose identity is in ``keys``."""
if not keys:
return set()
return {d["name"] for d in list_physical_disks() if d["key"] in keys}
+128 -2
View File
@@ -410,8 +410,133 @@ def _extract_temperature(data: dict[str, Any]) -> Optional[float]:
# ---------------------------------------------------------------------------
# ── Leaving idle and excluded disks alone ──────────────────────────
#
# `-n standby` keeps a periodic reader from waking a disk that is asleep.
# It does not let an awake one fall asleep: a drive's spin-down timer
# counts time without commands, and a read every minute resets it, so an
# unused drive whose timer is longer than that never spins down — and a
# drive that parks its heads when idle loads them again on every read.
#
# So a rotational disk with no I/O since it was last looked at is not read
# at all, not even asked for its power mode. The counters come from
# /proc/diskstats, which costs no disk access, and a SMART query does not
# move them — passthrough commands are not accounted as reads or writes —
# so any change means something else used the disk. A disk in use is
# already awake, and reading it then costs nothing. Solid-state disks have
# no spindle and no heads, and keep their reading.
#
# A disk seen for the first time is read once, so a Monitor that has just
# started still has values to show; after that it is left alone for as
# long as nothing uses it.
READ = "read"
IDLE = "idle"
EXCLUDED = "excluded"
_last_io: dict[str, tuple[int, int]] = {}
_idle_state: dict[str, float] = {}
_IDLE_TTL = 600 # same horizon as the standby badge
def _read_diskstats() -> dict[str, tuple[int, int]]:
"""Reads and writes completed per device, from /proc/diskstats."""
out: dict[str, tuple[int, int]] = {}
try:
with open("/proc/diskstats") as f:
for line in f:
parts = line.split()
if len(parts) < 8:
continue
try:
out[parts[2]] = (int(parts[3]), int(parts[7]))
except ValueError:
continue
except OSError:
pass
return out
def _is_rotational(disk_name: str) -> bool:
try:
with open(f"/sys/block/{disk_name}/queue/rotational") as f:
return f.read().strip() == "1"
except OSError:
return False
_EXCLUDED_TTL = 15
_excluded_cache: Optional[tuple[float, set]] = None
def excluded_disk_names() -> set:
"""Kernel names of the disks the user excluded. Fails open: if the
list cannot be read, nothing is excluded rather than everything.
Held for a few seconds, since every reader asks for every disk."""
global _excluded_cache
now = time.time()
with _cache_lock:
if _excluded_cache is not None and _excluded_cache[0] > now:
return set(_excluded_cache[1])
names: set = set()
try:
from health_persistence import health_persistence
keys = health_persistence.get_excluded_disk_keys()
if keys:
from disk_identity import names_for_keys
names = names_for_keys(keys)
except Exception:
names = set()
with _cache_lock:
_excluded_cache = (now + _EXCLUDED_TTL, set(names))
return names
def invalidate_disk_exclusions() -> None:
"""Apply a change to the exclusion list on the next read."""
global _excluded_cache
with _cache_lock:
_excluded_cache = None
_excluded_disk_names = excluded_disk_names
def disk_read_policy(disk_name: str, excluded: Optional[set] = None) -> str:
"""Whether a periodic reader may touch this disk now: READ, IDLE or
EXCLUDED. Shared by every reader that runs on its own, so the
temperature poller and the storage view cannot disagree about a disk.
``excluded`` lets a caller that checks many disks pass the list once."""
if excluded is None:
excluded = _excluded_disk_names()
if disk_name in excluded:
_idle_state.pop(disk_name, None)
return EXCLUDED
if not _is_rotational(disk_name):
return READ
current = _read_diskstats().get(disk_name)
with _cache_lock:
previous = _last_io.get(disk_name)
if current is not None:
_last_io[disk_name] = current
if current is None or previous is None or current != previous:
_idle_state.pop(disk_name, None)
return READ
_idle_state[disk_name] = time.time()
return IDLE
def is_disk_idle(disk_name: str) -> bool:
"""True while the disk is being left alone for having no I/O."""
ts = _idle_state.get(disk_name)
return ts is not None and (time.time() - ts) < _IDLE_TTL
def record_all_disk_temperatures() -> int:
"""Sample every non-USB disk and persist its temperature.
"""Sample the disks that may be read now and persist their temperature.
USB disks are included. A disk the user excluded, or a rotational one
with no I/O since the last cycle, is skipped — see ``disk_read_policy``.
Sampling fans out across a thread pool so a host with N disks pays
roughly the time of the slowest single ``smartctl`` call instead of
@@ -419,7 +544,8 @@ def record_all_disk_temperatures() -> int:
threading is enough — no need for asyncio. Returns the number of
rows actually written.
"""
disks = _list_target_disks()
excluded = _excluded_disk_names()
disks = [d for d in _list_target_disks() if disk_read_policy(d, excluded) == READ]
if not disks:
return 0
now = int(time.time())
+17 -10
View File
@@ -309,22 +309,29 @@ def accept_exception():
finding.get('incomplete') or not finding.get('scope')):
return jsonify(success=False, message="This finding cannot be accepted"), 400
expires_at = None
days = data.get('expires_in_days')
if days is not None:
try:
if isinstance(days, bool) or int(days) != float(days) or not 1 <= int(days) <= 3650:
raise ValueError("invalid expiry")
expires_at = int(time.time()) + int(days) * 86400
except (TypeError, ValueError):
return jsonify({"success": False,
"message": "Invalid expiry"}), 400
def _in_days(value, label):
"""A day count from now, or None. Same bounds as the expiry so a
reminder cannot be set further out than a decision can last."""
if value is None:
return None
if isinstance(value, bool) or int(value) != float(value) or not 1 <= int(value) <= 3650:
raise ValueError(f"invalid {label}")
return int(time.time()) + int(value) * 86400
try:
expires_at = _in_days(data.get('expires_in_days'), 'expiry')
# Independent of the expiry: it brings the decision back to the
# reader on that date without withdrawing it.
review_at = _in_days(data.get('review_in_days'), 'review date')
except (TypeError, ValueError) as e:
return jsonify({"success": False, "message": str(e)}), 400
audit_store.accept_risk(
check_id, reason,
accepted_by=_actor(),
expires_at=expires_at,
scope=finding['scope'],
review_at=review_at,
)
return jsonify({"success": True})
except ValueError as e:
+116
View File
@@ -5,6 +5,7 @@ Flask routes for health monitoring with persistence support
from flask import Blueprint, jsonify, request
from health_monitor import health_monitor
from health_persistence import health_persistence
from jwt_middleware import require_auth, require_admin_scope
# Sprint 13: remote-mount monitor (NFS/CIFS/SMB) — separate module so a
# missing helper doesn't crash the health blueprint.
@@ -17,6 +18,7 @@ except ImportError:
health_bp = Blueprint('health', __name__)
@health_bp.route('/api/health/status', methods=['GET'])
@require_auth
def get_health_status():
"""Get overall health status summary"""
try:
@@ -26,6 +28,7 @@ def get_health_status():
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/details', methods=['GET'])
@require_auth
def get_health_details():
"""Get detailed health status with all checks"""
try:
@@ -58,6 +61,7 @@ def get_system_info():
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/acknowledge', methods=['POST'])
@require_admin_scope
def acknowledge_error():
"""
Acknowledge/dismiss an error manually.
@@ -156,6 +160,7 @@ def acknowledge_error():
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/un-acknowledge', methods=['POST'])
@require_admin_scope
def unacknowledge_error():
"""
Re-enable a previously dismissed error.
@@ -203,6 +208,7 @@ def unacknowledge_error():
@health_bp.route('/api/health/active-errors', methods=['GET'])
@require_auth
def get_active_errors():
"""Get all active persistent errors"""
try:
@@ -213,6 +219,7 @@ def get_active_errors():
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/dismissed', methods=['GET'])
@require_auth
def get_dismissed_errors():
"""
Get dismissed errors that are still within their suppression period.
@@ -225,6 +232,7 @@ def get_dismissed_errors():
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/full', methods=['GET'])
@require_auth
def get_full_health():
"""
Get complete health data in a single request: detailed status + active errors + dismissed.
@@ -271,6 +279,7 @@ def get_full_health():
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/cleanup-orphans', methods=['POST'])
@require_admin_scope
def cleanup_orphan_errors():
"""
Clean up errors for devices that no longer exist in the system.
@@ -331,6 +340,7 @@ def cleanup_orphan_errors():
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/pending-notifications', methods=['GET'])
@require_auth
def get_pending_notifications():
"""
Get events pending notification (for future Telegram/Gotify/Discord integration).
@@ -343,6 +353,7 @@ def get_pending_notifications():
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/mark-notified', methods=['POST'])
@require_admin_scope
def mark_events_notified():
"""
Mark events as notified after notification was sent successfully.
@@ -364,6 +375,7 @@ def mark_events_notified():
@health_bp.route('/api/health/settings', methods=['GET'])
@require_auth
def get_health_settings():
"""
Get per-category suppression duration settings.
@@ -377,6 +389,7 @@ def get_health_settings():
@health_bp.route('/api/health/settings', methods=['POST'])
@require_admin_scope
def save_health_settings():
"""
Save per-category suppression duration settings.
@@ -422,6 +435,7 @@ def save_health_settings():
# ── Remote Storage Exclusions Endpoints ──
@health_bp.route('/api/health/remote-storages', methods=['GET'])
@require_auth
def get_remote_storages():
"""
Get list of all remote storages with their exclusion status.
@@ -472,6 +486,7 @@ def get_remote_storages():
@health_bp.route('/api/health/storage-exclusions', methods=['GET'])
@require_auth
def get_storage_exclusions():
"""Get all storage exclusions."""
try:
@@ -482,6 +497,7 @@ def get_storage_exclusions():
@health_bp.route('/api/health/storage-exclusions', methods=['POST'])
@require_admin_scope
def save_storage_exclusion():
"""
Add or update a storage exclusion.
@@ -535,6 +551,7 @@ def save_storage_exclusion():
@health_bp.route('/api/health/storage-exclusions/<storage_name>', methods=['DELETE'])
@require_admin_scope
def delete_storage_exclusion(storage_name):
"""Remove a storage from the exclusion list."""
try:
@@ -555,6 +572,7 @@ def delete_storage_exclusion(storage_name):
# ═══════════════════════════════════════════════════════════════════════════
@health_bp.route('/api/health/interfaces', methods=['GET'])
@require_auth
def get_network_interfaces():
"""Get all network interfaces with their exclusion status."""
try:
@@ -615,6 +633,7 @@ def get_network_interfaces():
@health_bp.route('/api/health/interface-exclusions', methods=['GET'])
@require_auth
def get_interface_exclusions():
"""Get all interface exclusions."""
try:
@@ -625,6 +644,7 @@ def get_interface_exclusions():
@health_bp.route('/api/health/interface-exclusions', methods=['POST'])
@require_admin_scope
def save_interface_exclusion():
"""
Add or update an interface exclusion.
@@ -677,6 +697,7 @@ def save_interface_exclusion():
@health_bp.route('/api/health/interface-exclusions/<interface_name>', methods=['DELETE'])
@require_admin_scope
def delete_interface_exclusion(interface_name):
"""Remove an interface from the exclusion list."""
try:
@@ -692,7 +713,102 @@ def delete_interface_exclusion(interface_name):
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/disks', methods=['GET'])
@require_auth
def get_disks_for_exclusion():
"""Physical disks with whether each is excluded from periodic reads.
Listed from udev and /sys only, so opening the settings page does not
touch a disk the user is about to exclude precisely to leave it alone.
"""
try:
from disk_identity import list_physical_disks
import disk_temperature_history as _dth
excluded = {e['disk_key']: e for e in health_persistence.get_excluded_disks()}
present = set()
result = []
for disk in list_physical_disks():
present.add(disk['key'])
entry = excluded.get(disk['key'])
result.append({
**disk,
'excluded': entry is not None,
'excluded_at': entry.get('excluded_at') if entry else None,
'idle': _dth.is_disk_idle(disk['name']),
'present': True,
})
# An excluded disk that is not connected right now — an unplugged USB
# drive — stays in the list, so its exclusion can still be seen and
# removed rather than silently waiting for it to come back.
for key, entry in excluded.items():
if key in present:
continue
result.append({
'name': entry.get('disk_name') or '',
'key': key,
'model': entry.get('model') or '',
'serial': entry.get('serial') or '',
'size_bytes': 0,
'transport': '',
'rotational': False,
'excluded': True,
'excluded_at': entry.get('excluded_at'),
'idle': False,
'present': False,
})
result.sort(key=lambda d: (not d['present'], d['name']))
return jsonify({'disks': result})
except Exception as e:
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/disk-exclusions', methods=['POST'])
@require_admin_scope
def save_disk_exclusion():
"""Exclude a disk from periodic reads.
Request body: {"disk_key": "serial:WD-...", "disk_name": "sdb",
"model": "...", "serial": "...", "reason": "..."}
The key is the one /api/health/disks reports; it follows the disk
across kernel renames.
"""
try:
data = request.get_json(silent=True) or {}
disk_key = str(data.get('disk_key') or '').strip()
if not disk_key or ':' not in disk_key or len(disk_key) > 200:
return jsonify({'error': 'a valid disk_key is required'}), 400
ok = health_persistence.exclude_disk(
disk_key,
disk_name=str(data.get('disk_name') or '')[:64] or None,
model=str(data.get('model') or '')[:128] or None,
serial=str(data.get('serial') or '')[:128] or None,
reason=str(data.get('reason') or '')[:500] or None,
)
if not ok:
return jsonify({'error': 'Failed to save exclusion'}), 500
import disk_temperature_history as _dth
_dth.invalidate_disk_exclusions()
return jsonify({'success': True, 'disk_key': disk_key})
except Exception as e:
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/health/disk-exclusions/<path:disk_key>', methods=['DELETE'])
@require_admin_scope
def delete_disk_exclusion(disk_key):
"""Put a disk back under periodic reads."""
try:
if not health_persistence.remove_disk_exclusion(disk_key):
return jsonify({'error': 'Disk not found in exclusions'}), 404
import disk_temperature_history as _dth
_dth.invalidate_disk_exclusions()
return jsonify({'success': True, 'disk_key': disk_key})
except Exception as e:
return jsonify({'error': str(e)}), 500
@health_bp.route('/api/mounts', methods=['GET'])
@require_auth
def get_remote_mounts():
"""Sprint 13: list NFS/CIFS/SMB mounts on the host AND inside every
running LXC, with per-mount health (reachable / stale / read-only).
+64 -11
View File
@@ -1616,8 +1616,10 @@ _system_info_cache = {
'proxmox_version_time': 0,
'available_updates': 0,
'available_updates_time': 0,
'available_updates_stamp': 0.0,
}
_SYSTEM_INFO_CACHE_TTL = 21600 # 6 hours - update notifications are sent once per 24h
_AVAILABLE_UPDATES_MIN_INTERVAL = 30 # seconds between apt recounts when apt state moves
# Cache for pvesh cluster resources (reduces repeated API calls)
_pvesh_cache = {
@@ -3316,15 +3318,39 @@ def get_proxmox_version():
_system_info_cache['proxmox_version_time'] = now
return proxmox_version
def _apt_state_stamp():
"""Newest mtime of the files that decide what `apt list --upgradable`
answers: dpkg's status file (what is installed) and apt's package
lists (what is on offer). Any upgrade moves it — whichever way the
packages were installed."""
newest = 0.0
for path in ('/var/lib/dpkg/status', '/var/lib/apt/lists', '/var/cache/apt/pkgcache.bin'):
try:
newest = max(newest, os.path.getmtime(path))
except OSError:
pass
return newest
def get_available_updates():
"""Get the number of available package updates. Cached for 6 hours."""
"""Get the number of available package updates. Cached for 6 hours,
or until apt's own state moves — an upgrade that finishes two minutes
after the count was taken must not leave the overview showing what
was pending before it ran."""
global _system_info_cache
now = time.time()
if _system_info_cache['available_updates_time'] > 0 and \
now - _system_info_cache['available_updates_time'] < _SYSTEM_INFO_CACHE_TTL:
return _system_info_cache['available_updates']
stamp = _apt_state_stamp()
age = now - _system_info_cache['available_updates_time']
if _system_info_cache['available_updates_time'] > 0:
# dpkg rewrites its status file once per package, so during an
# upgrade the stamp moves with every one of them. The floor keeps
# that from turning each overview poll into an apt call.
if age < _AVAILABLE_UPDATES_MIN_INTERVAL:
return _system_info_cache['available_updates']
if stamp == _system_info_cache['available_updates_stamp'] and age < _SYSTEM_INFO_CACHE_TTL:
return _system_info_cache['available_updates']
available_updates = 0
try:
# Use apt list --upgradable to count available updates
@@ -3340,6 +3366,7 @@ def get_available_updates():
_system_info_cache['available_updates'] = available_updates
_system_info_cache['available_updates_time'] = now
_system_info_cache['available_updates_stamp'] = stamp
return available_updates
# AGREGANDO FUNCIÓN PARA PARSEAR PROCESOS DE INTEL_GPU_TOP (SIN -J)
@@ -4123,9 +4150,13 @@ def get_storage_info():
# temperature graph isn't a monitor bug — the
# disk is parked. See issue #232.
in_standby = False
in_idle = False
in_excluded = False
try:
import disk_temperature_history as _dth
in_standby = _dth.is_disk_in_standby(disk_name)
in_idle = _dth.is_disk_idle(disk_name)
in_excluded = disk_name in _dth.excluded_disk_names()
except Exception:
pass
physical_disks[disk_name] = {
@@ -4135,6 +4166,8 @@ def get_storage_info():
'size_bytes': disk_size_bytes,
'temperature': smart_data.get('temperature', 0),
'standby': in_standby,
'idle': in_idle,
'excluded': in_excluded,
'health': smart_data.get('health', 'unknown'),
'power_on_hours': smart_data.get('power_on_hours', 0),
'smart_status': smart_data.get('smart_status', 'unknown'),
@@ -4860,6 +4893,22 @@ def get_smart_data(disk_name):
if cached and now - cached[0] < _SMART_RESULT_TTL:
return dict(cached[1])
# Excluded, or rotational with no I/O since it was last looked at: send
# it nothing — not even the power-mode question below, which is still a
# command. Serve what is known, without a temperature that would only be
# stale. Same rule as the temperature poller, so the two agree.
try:
import disk_temperature_history as _dth
policy = _dth.disk_read_policy(disk_name)
except Exception:
policy = 'read'
if policy != 'read':
base = dict(cached[1]) if cached else _smart_default_payload()
base['temperature'] = 0
base['excluded'] = policy == 'excluded'
base['idle'] = policy == 'idle'
return base
if _hdd_in_standby(disk_name):
# Keep serving the last known values (temperature blanked, since
# we don't have a fresh one) so the card stays populated while
@@ -14433,7 +14482,7 @@ def api_health_thresholds_get():
@app.route('/api/health/thresholds', methods=['PUT'])
@require_auth
@require_admin_scope
def api_health_thresholds_put():
"""Save a partial threshold payload. Body shape mirrors DEFAULTS
but the leaves are bare numbers, not metadata dicts. Sections not
@@ -14453,7 +14502,7 @@ def api_health_thresholds_put():
@app.route('/api/health/thresholds/reset', methods=['POST'])
@require_auth
@require_admin_scope
def api_health_thresholds_reset():
"""Reset thresholds. ?section=<name> resets one section, no
parameter resets everything to recommended."""
@@ -14473,7 +14522,7 @@ def api_health_thresholds_reset():
@app.route('/api/health/acknowledge', methods=['POST'])
@require_auth
@require_admin_scope
def api_health_acknowledge():
"""Acknowledge/dismiss a health error by error_key.
@@ -14502,7 +14551,7 @@ def api_health_acknowledge():
@app.route('/api/health/un-acknowledge', methods=['POST'])
@require_auth
@require_admin_scope
def api_health_unacknowledge():
"""Reverse a previous dismiss — re-enables the alert so it can fire again.
@@ -20148,7 +20197,11 @@ def _borg_env_for(target: dict, extra: dict | None = None) -> dict:
env['BORG_PASSPHRASE'] = pw
ssh_key = target.get('ssh_key') or ''
if ssh_key:
env['BORG_RSH'] = f'ssh -i {ssh_key} -o StrictHostKeyChecking=accept-new'
# IdentitiesOnly keeps ssh from offering root's default keys first:
# a server that only accepts the ProxMenux key can hit MaxAuthTries
# before it is ever tried. borg adds `-p <port>` from the ssh:// URL.
env['BORG_RSH'] = (f'ssh -i {ssh_key} -o IdentitiesOnly=yes '
'-o StrictHostKeyChecking=accept-new')
# Non-interactive: if borg would prompt about a relocated repo, take
# the safe answer instead of hanging the request.
env['BORG_RELOCATED_REPO_ACCESS_IS_OK'] = 'yes'
+78 -1
View File
@@ -421,6 +421,22 @@ class HealthPersistence:
)
''')
cursor.execute('CREATE INDEX IF NOT EXISTS idx_excluded_interface ON excluded_interfaces(interface_name)')
# Disks the user wants left alone: no periodic SMART or temperature
# reads, so a drive can reach its own spin-down and stop cycling its
# heads. Keyed by a stable identity rather than the kernel name, which
# a USB drive can change (sda -> sdb) on every reconnection.
cursor.execute('''
CREATE TABLE IF NOT EXISTS excluded_disks (
id INTEGER PRIMARY KEY AUTOINCREMENT,
disk_key TEXT UNIQUE NOT NULL,
disk_name TEXT,
model TEXT,
serial TEXT,
excluded_at TEXT NOT NULL,
reason TEXT
)
''')
conn.commit()
@@ -430,7 +446,7 @@ class HealthPersistence:
required_tables = {'errors', 'events', 'system_capabilities', 'user_settings',
'notification_history', 'notification_last_sent', 'notification_delivery_claims',
'disk_registry', 'disk_observations',
'excluded_storages', 'excluded_interfaces'}
'excluded_storages', 'excluded_interfaces', 'excluded_disks'}
missing = required_tables - tables
if missing:
print(f"[HealthPersistence] WARNING: Missing tables after init: {missing}")
@@ -3257,6 +3273,67 @@ class HealthPersistence:
print(f"[HealthPersistence] Error removing interface exclusion: {e}")
return False
# ------------------------------------------------------------------
# Disk exclusions
# ------------------------------------------------------------------
def get_excluded_disks(self) -> List[Dict[str, Any]]:
"""Every disk the user has excluded from periodic reads."""
try:
with self._db_connection(row_factory=True) as conn:
cursor = conn.cursor()
cursor.execute('''
SELECT disk_key, disk_name, model, serial, excluded_at, reason
FROM excluded_disks
''')
return [dict(row) for row in cursor.fetchall()]
except Exception as e:
print(f"[HealthPersistence] Error getting excluded disks: {e}")
return []
def exclude_disk(self, disk_key: str, disk_name: str = None, model: str = None,
serial: str = None, reason: str = None) -> bool:
"""Add a disk to the exclusion list, or refresh its display fields."""
try:
with self._db_connection() as conn:
cursor = conn.cursor()
cursor.execute('''
INSERT INTO excluded_disks
(disk_key, disk_name, model, serial, excluded_at, reason)
VALUES (?, ?, ?, ?, ?, ?)
ON CONFLICT(disk_key) DO UPDATE SET
disk_name = excluded.disk_name,
model = excluded.model,
serial = excluded.serial
''', (disk_key, disk_name, model, serial, datetime.now().isoformat(), reason))
conn.commit()
return True
except Exception as e:
print(f"[HealthPersistence] Error excluding disk: {e}")
return False
def remove_disk_exclusion(self, disk_key: str) -> bool:
"""Put a disk back under periodic reads."""
try:
with self._db_connection() as conn:
cursor = conn.cursor()
cursor.execute('DELETE FROM excluded_disks WHERE disk_key = ?', (disk_key,))
conn.commit()
return cursor.rowcount > 0
except Exception as e:
print(f"[HealthPersistence] Error removing disk exclusion: {e}")
return False
def get_excluded_disk_keys(self) -> set:
"""Stable keys of the excluded disks (see disk_identity.disk_key)."""
try:
with self._db_connection() as conn:
cursor = conn.cursor()
cursor.execute('SELECT disk_key FROM excluded_disks')
return {row[0] for row in cursor.fetchall()}
except Exception:
return set()
def get_excluded_interface_names(self, check_type: str = 'health') -> set:
"""
Get set of interface names excluded for a specific check type.
+70 -2
View File
@@ -2622,6 +2622,7 @@ def annotate_delegated_apps(apps: list, docker_inventory: dict) -> None:
# showing the version of an image it no longer runs.
app['docker_available_version'] = None
app['docker_update_available'] = None
app['docker_pinned'] = None
link = resolve_docker_image_for_app(app, docker_inventory)
app['docker_image_reference'] = link.get('image_reference')
app['docker_binding_error'] = link.get('error')
@@ -2633,6 +2634,7 @@ def annotate_delegated_apps(apps: list, docker_inventory: dict) -> None:
continue
app['docker_available_version'] = image.get('available_version')
app['docker_update_available'] = image.get('update_available')
app['docker_pinned'] = image.get('pinned')
break
except Exception:
pass
@@ -2903,6 +2905,7 @@ def _docker_inventory_from_ct(vmid) -> dict:
"architecture": str(inspected_image.get("Architecture") or ""),
"variant": str(inspected_image.get("Variant") or ""),
},
"pinned": False,
"available_version": None,
"available_version_source": None,
"update_available": None,
@@ -2911,13 +2914,78 @@ def _docker_inventory_from_ct(vmid) -> dict:
if len(images) >= _DOCKER_MAX_IMAGES:
break
# A container pinned by digest runs exactly the image it names; a newer
# tag upstream does not move it, only an edit to its reference does. It is
# listed under that reference with its installed version, and is never
# compared with the registry nor offered an update.
pinned_groups: dict[str, list[dict]] = {}
for item in containers:
reference = str(item.get("image_reference") or item.get("image") or "").strip()
if "@" in reference:
pinned_groups.setdefault(reference, []).append(item)
for reference in sorted(pinned_groups):
if len(images) >= _DOCKER_MAX_IMAGES:
break
name, _, pinned_digest = reference.partition("@")
if not re.fullmatch(r"sha256:[0-9a-f]{64}", pinned_digest):
continue
final_component = name.rsplit("/", 1)[-1]
repository, tag = name.rsplit(":", 1) if ":" in final_component else (name, "")
parsed = _parse_docker_reference(repository, tag or pinned_digest)
if not parsed or reference in seen:
continue
seen.add(reference)
parsed = {**parsed, "tag": tag, "reference": reference}
group = pinned_groups[reference]
image_id = next((str(item.get("image_id")) for item in group if item.get("image_id")), "")
inspected_image = (
inspected_images.get(image_id)
or inspected_images.get(image_id.removeprefix("sha256:"))
or {}
)
installed_version, installed_version_source = _docker_version_from_image_inspect(
parsed, inspected_image,
)
primary_compose = group[0].get("compose") or {}
display_meta = _docker_service_catalog_meta(
str(primary_compose.get("service") or ""),
str(group[0].get("name") or ""),
reference,
)
images.append({
**parsed,
"local_digest": pinned_digest,
"remote_digest": None,
"image_id": image_id,
"used_by": sorted({item["name"] for item in group}),
"update_targets": [],
"standalone_containers": [],
"display_name": display_meta.get("name"),
"logo_url": display_meta.get("logo_url"),
"installed_version": installed_version,
"installed_version_source": installed_version_source,
"platform": {
"os": str(inspected_image.get("Os") or ""),
"architecture": str(inspected_image.get("Architecture") or ""),
"variant": str(inspected_image.get("Variant") or ""),
},
"pinned": True,
"available_version": None,
"available_version_source": None,
"update_available": None,
"error": None,
})
def _check(item: dict) -> tuple[str, Optional[str], Optional[str]]:
remote, error = _fetch_registry_manifest_digest(item)
return item["reference"], remote, error
if images:
with concurrent.futures.ThreadPoolExecutor(max_workers=min(4, len(images))) as pool:
results = list(pool.map(_check, images))
checkable = [item for item in images if not item.get("pinned")]
results = []
if checkable:
with concurrent.futures.ThreadPoolExecutor(max_workers=min(4, len(checkable))) as pool:
results = list(pool.map(_check, checkable))
by_ref = {ref: (digest, error) for ref, digest, error in results}
for item in images:
remote, remote_error = by_ref.get(item["reference"], (None, None))
+9 -8
View File
@@ -40,9 +40,10 @@ CATALOG_FILE = os.path.join(OCI_BASE_DIR, "catalog.json")
INSTALLED_FILE = os.path.join(OCI_BASE_DIR, "installed.json")
INSTANCES_DIR = os.path.join(OCI_BASE_DIR, "instances")
# Source catalog from Scripts (bundled with ProxMenux)
SCRIPTS_CATALOG = "/usr/local/share/proxmenux/scripts/oci/catalog.json"
DEV_SCRIPTS_CATALOG = os.path.join(os.path.dirname(__file__), "..", "..", "Scripts", "oci", "catalog.json")
# Source catalog shipped with ProxMenux, inside the OCI engine
SCRIPTS_CATALOG = os.path.join(OCI_BASE_DIR, "engine", "addons", "secure-gateway.json")
LEGACY_SCRIPTS_CATALOG = "/usr/local/share/proxmenux/scripts/oci/catalog.json"
DEV_SCRIPTS_CATALOG = os.path.join(os.path.dirname(__file__), "..", "..", "oci", "addons", "secure-gateway.json")
# Encryption key file
ENCRYPTION_KEY_FILE = os.path.join(OCI_BASE_DIR, ".encryption_key")
@@ -143,10 +144,10 @@ def ensure_oci_directories():
os.makedirs(INSTANCES_DIR, exist_ok=True)
if not os.path.exists(CATALOG_FILE):
if os.path.exists(SCRIPTS_CATALOG):
shutil.copy2(SCRIPTS_CATALOG, CATALOG_FILE)
elif os.path.exists(DEV_SCRIPTS_CATALOG):
shutil.copy2(DEV_SCRIPTS_CATALOG, CATALOG_FILE)
for source in (SCRIPTS_CATALOG, LEGACY_SCRIPTS_CATALOG, DEV_SCRIPTS_CATALOG):
if os.path.exists(source):
shutil.copy2(source, CATALOG_FILE)
break
if not os.path.exists(INSTALLED_FILE):
with open(INSTALLED_FILE, 'w') as f:
@@ -689,7 +690,7 @@ def load_catalog() -> Dict[str, Any]:
"""Load the OCI app catalog."""
ensure_oci_directories()
for path in [CATALOG_FILE, SCRIPTS_CATALOG, DEV_SCRIPTS_CATALOG]:
for path in [CATALOG_FILE, SCRIPTS_CATALOG, LEGACY_SCRIPTS_CATALOG, DEV_SCRIPTS_CATALOG]:
if os.path.exists(path):
try:
with open(path, 'r') as f: