From 1bef36fb2041cbb4f6d67136b7ca9a5b940df782 Mon Sep 17 00:00:00 2001 From: MacRimi Date: Wed, 2 Sep 2026 12:05:53 +0200 Subject: [PATCH] drop live AI catalog refresh + GH Action --- .../scripts/ai-models-verifier/.env.example | 16 -- .github/scripts/ai-models-verifier/README.md | 37 --- .github/scripts/ai-models-verifier/apply.py | 143 ---------- .github/scripts/ai-models-verifier/prompts.py | 50 ---- .../scripts/ai-models-verifier/providers.py | 263 ------------------ .github/scripts/ai-models-verifier/verify.py | 232 --------------- .github/workflows/verify-ai-models.yml | 103 ------- AppImage/components/notification-settings.tsx | 71 +---- AppImage/config/verified_ai_models.json | 138 ++++----- AppImage/messages/de/common.json | 6 +- AppImage/messages/en/common.json | 6 +- AppImage/messages/es/common.json | 6 +- AppImage/messages/fr/common.json | 6 +- AppImage/messages/it/common.json | 6 +- AppImage/messages/pt/common.json | 6 +- AppImage/messages/sk/common.json | 6 +- AppImage/messages/sv/common.json | 6 +- AppImage/scripts/flask_notification_routes.py | 261 +++-------------- 18 files changed, 101 insertions(+), 1261 deletions(-) delete mode 100644 .github/scripts/ai-models-verifier/.env.example delete mode 100644 .github/scripts/ai-models-verifier/README.md delete mode 100644 .github/scripts/ai-models-verifier/apply.py delete mode 100644 .github/scripts/ai-models-verifier/prompts.py delete mode 100644 .github/scripts/ai-models-verifier/providers.py delete mode 100755 .github/scripts/ai-models-verifier/verify.py delete mode 100644 .github/workflows/verify-ai-models.yml diff --git a/.github/scripts/ai-models-verifier/.env.example b/.github/scripts/ai-models-verifier/.env.example deleted file mode 100644 index 552f7180..00000000 --- a/.github/scripts/ai-models-verifier/.env.example +++ /dev/null @@ -1,16 +0,0 @@ -# ProxMenux AI-model verifier — API keys -# Copy this file to .env (gitignored) and fill only the ones you have. - -OPENAI_API_KEY= -# Optional: for LiteLLM/MLX/LM Studio/vLLM/LocalAI/Ollama-proxy testing, -# point OPENAI_API_KEY to any non-empty placeholder and set OPENAI_BASE_URL -# to the endpoint's root (without /v1 — the tool appends it). -# OPENAI_BASE_URL=http://localhost:4000 - -GROQ_API_KEY= - -GEMINI_API_KEY= - -ANTHROPIC_API_KEY= - -OPENROUTER_API_KEY= diff --git a/.github/scripts/ai-models-verifier/README.md b/.github/scripts/ai-models-verifier/README.md deleted file mode 100644 index 14430aee..00000000 --- a/.github/scripts/ai-models-verifier/README.md +++ /dev/null @@ -1,37 +0,0 @@ -# AI models verifier (public copy) - -Standalone verifier used by the daily GitHub Action to refresh -`AppImage/config/verified_ai_models.json`. - -The code lives here so the Action can execute it. API keys are read from -GitHub Secrets at run time and never written to disk. - -Local dev runs (interactive verifier over your own keys) can keep using -the private copy — `verify.py` is identical. - -## What the Action does - -Each run: - -1. Loads keys from Secrets into environment variables. -2. Runs `verify.py --json-out /tmp/report.json` against every provider that - has a key set. -3. Rewrites `AppImage/config/verified_ai_models.json` with the passing - models, sorted with the recommended one first per provider. -4. Bumps the `_updated` field to the current date. -5. If the file changed, commits directly to `main` as a bot commit. - -## Adding provider keys - -- Repository → Settings → Secrets and variables → Actions. -- Add each key with the exact name expected by `verify.py`: - `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GROQ_API_KEY`, `GEMINI_API_KEY`, - `OPENROUTER_API_KEY`. -- Any provider without a key is silently skipped — the Action logs a - warning and continues with the rest. - -## Running the Action on demand - -The workflow accepts `workflow_dispatch`, so you can trigger a refresh -manually from the Actions tab. Useful when a new model has just been -released upstream and you don't want to wait for the daily cron. diff --git a/.github/scripts/ai-models-verifier/apply.py b/.github/scripts/ai-models-verifier/apply.py deleted file mode 100644 index c5dff76b..00000000 --- a/.github/scripts/ai-models-verifier/apply.py +++ /dev/null @@ -1,143 +0,0 @@ -#!/usr/bin/env python3 -"""Apply a verifier report to ``AppImage/config/verified_ai_models.json``. - -Reads the machine-readable report emitted by ``verify.py --json-out`` and -merges the passing models into the on-disk catalog. Preserves the -maintainer's editorial curation across three axes: - -* ``_exclude``: per-provider list of model IDs (exact match) that must - never appear in the surfaced ``models`` list even when the verifier - passes them. Meant for models that respond correctly to the technical - test but are the wrong fit for notification translation — Arabic-only - bases, Chinese-first fine-tunes, safety-classifier variants, - agentic-only endpoints, legacy dated snapshots, etc. -* ``recommended``: if the current recommendation is still in the - passing (and non-excluded) set, it is preserved. Only when the - previous recommendation disappears (deprecated upstream, or newly - excluded) is a fallback chosen — the fastest passing model. -* ``_note`` / ``_deprecated``: never touched. Those are maintainer - annotations that outlive any single verifier run. - -Fail-safe rules: -* Providers absent from the report (no API key configured in the - Action for that run) are left untouched. -* Providers whose report carries an error are left untouched. -* If the ``_exclude`` filter drops every passing model, the block is - left untouched — an empty models list would silently kill the - provider in the UI; keeping the previous list is more forgiving - than shipping "nothing works". -* ``_updated`` bumps to today's date only when the merge actually - changed something. A no-op run leaves the file byte-identical. - -Exits 0 when the file is unchanged, 10 when it was updated. The -workflow uses that exit code to decide whether to commit. -""" -from __future__ import annotations - -import argparse -import datetime as dt -import fnmatch -import json -import sys -from pathlib import Path - - -def _load_json(path: Path) -> dict: - with open(path, "r", encoding="utf-8") as fh: - return json.load(fh) - - -def _save_json(path: Path, data: dict) -> None: - tmp = path.with_suffix(path.suffix + ".tmp") - with open(tmp, "w", encoding="utf-8") as fh: - json.dump(data, fh, indent=2, ensure_ascii=False) - fh.write("\n") - tmp.replace(path) - - -def _is_excluded(model: str, patterns: list[str]) -> bool: - """Match a model against the ``_exclude`` list. Supports exact - matches and shell-style globs (``gpt-4o-*``, ``*-2024-*``, ...) so - a provider that periodically publishes dated snapshots can be - covered by a single pattern instead of one entry per date.""" - for pat in patterns: - if pat == model or fnmatch.fnmatchcase(model, pat): - return True - return False - - -def _passing_models(provider_report: dict, exclude: list[str]) -> list[str]: - """Passing models minus the editorial exclusion list, fastest first.""" - passing = [ - r for r in provider_report.get("results", []) - if r.get("verdict") == "pass" and not _is_excluded(r.get("model", ""), exclude) - ] - passing.sort(key=lambda r: r.get("latency_s", 999)) - return [r["model"] for r in passing] - - -def apply_report(report_path: Path, catalog_path: Path, today: str) -> bool: - """Rewrite the catalog from the report. Returns True if it changed.""" - report = _load_json(report_path) - catalog = _load_json(catalog_path) if catalog_path.exists() else {} - - changed = False - for provider_report in report: - name = provider_report.get("provider") - if not name: - continue - if provider_report.get("error"): - print(f"[{name}] skipped — verifier reported error: {provider_report['error']}", - file=sys.stderr) - continue - - block = catalog.setdefault(name, {}) - exclude = list(block.get("_exclude", [])) - passing = _passing_models(provider_report, exclude) - - if not passing: - # Either the verifier returned no passes for this provider, - # or every pass got filtered by _exclude. Both cases mean - # "no signal we can trust to overwrite the curated list"; - # leaving the block alone is safer than blanking it. - print(f"[{name}] skipped — no passing models after exclude filter", - file=sys.stderr) - continue - - prev_models = list(block.get("models", [])) - prev_recommended = block.get("recommended", "") - # Preserve the maintainer's choice of recommended when it is - # still valid. Only fall back to fastest when the previous - # value disappeared from the passing set. - recommended = prev_recommended if prev_recommended in passing else passing[0] - - if sorted(prev_models) != sorted(passing) or prev_recommended != recommended: - block["models"] = passing - block["recommended"] = recommended - changed = True - print(f"[{name}] updated — {len(passing)} models, recommended={recommended}") - else: - print(f"[{name}] unchanged — {len(passing)} models") - - if changed: - catalog["_updated"] = today - _save_json(catalog_path, catalog) - return changed - - -def main() -> int: - ap = argparse.ArgumentParser(description=__doc__) - ap.add_argument("--report", required=True, help="verify.py --json-out path") - ap.add_argument("--catalog", required=True, - help="AppImage/config/verified_ai_models.json path") - ap.add_argument("--today", default=None, - help="Override the date written into _updated (YYYY-MM-DD).") - args = ap.parse_args() - - today = args.today or dt.datetime.utcnow().strftime("%Y-%m-%d") - changed = apply_report(Path(args.report), Path(args.catalog), today) - return 10 if changed else 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/.github/scripts/ai-models-verifier/prompts.py b/.github/scripts/ai-models-verifier/prompts.py deleted file mode 100644 index 77c3ccc9..00000000 --- a/.github/scripts/ai-models-verifier/prompts.py +++ /dev/null @@ -1,50 +0,0 @@ -"""Standardized test prompt for ProxMenux AI-model verification. - -Mirrors the real AI-enrichment use case: take a raw Proxmox system -notification (English, with technical identifiers), translate it into -Spanish, explain in plain terms, and suggest one concrete action. It is -intentionally simple — if a model can't do this, it won't do the real -thing either. Models that pass this test are fine for inclusion in -verified_ai_models.json. -""" - -SYSTEM_PROMPT = ( - "You are a Proxmox system-notification assistant. " - "When given a raw notification from a Proxmox host, you: " - "(1) translate it into Spanish, " - "(2) explain in 2-3 sentences what the user is seeing and the likely cause, " - "(3) suggest ONE concrete next action. " - "Keep technical identifiers (device paths like /dev/sdd, SMART keywords, " - "ata port numbers, BDFs) in their original form. " - "Respond only in Spanish. Stay under 200 tokens total." -) - -# Realistic ProxMenux notification payload: multi-line body with -# SMART/ATA vocabulary and a frequency hint — the exact shape the real -# pipeline emits. -USER_MESSAGE = ( - "Event: disk_io_error\n" - "Severity: CRITICAL\n" - "Host: pve-constructor\n" - "Device: /dev/sdd\n" - "SMART status: PASSED\n" - "Summary: 3 I/O event(s) in 5 minutes, disk passed SMART short test\n" - "Sample kernel line: ata4.00: exception Emask 0x0 SAct 0x804000 SErr 0x0 action 0x6\n" - "Frequency: 3 occurrences in 24h, first seen 6h ago" -) - -# Common Spanish stopwords. A response missing ALL of these is almost -# certainly not Spanish (or empty/truncated). Cheap heuristic, good -# enough for a coarse pass/fail. -REQUIRED_SPANISH_HINTS = [ - " el ", " la ", " los ", " las ", " un ", " una ", - " de ", " del ", " en ", " con ", " que ", " para ", - " es ", " se ", " ha ", " por ", " y ", -] - -# Domain keywords — at least one must appear to confirm the model -# actually engaged with the notification instead of replying generically. -DOMAIN_HINTS = [ - "disco", "sdd", "smart", "ata", "i/o", "e/s", "error", - "kernel", "proxmox", -] diff --git a/.github/scripts/ai-models-verifier/providers.py b/.github/scripts/ai-models-verifier/providers.py deleted file mode 100644 index 5e13ea67..00000000 --- a/.github/scripts/ai-models-verifier/providers.py +++ /dev/null @@ -1,263 +0,0 @@ -"""Self-contained API wrappers for AI-model verification. - -Kept independent from the ProxMenux AppImage's ai_providers module so -this tool can live in a private repo with no import coupling to the -public project. Uses only the Python standard library. -""" -from __future__ import annotations - -import json -import urllib.error -import urllib.request -from typing import List, Optional - - -class ProviderError(Exception): - pass - - -class Provider: - """Base class. Subclasses implement list_models() and generate().""" - name = "base" - - def __init__(self, api_key: str, base_url: Optional[str] = None): - self.api_key = api_key - self.base_url = base_url - - def list_models(self) -> List[str]: - raise NotImplementedError - - def generate(self, model: str, system: str, user: str, - max_tokens: int = 250, timeout: int = 30) -> str: - raise NotImplementedError - - # ── HTTP helpers ──────────────────────────────────────────── - - # Cloudflare in front of api.groq.com (and probably other providers - # over time) returns 403 "error code: 1010" for the default - # `Python-urllib/3.x` User-Agent — the "browser signature ban" rule. - # A plain identifier is enough to get through; we're not spoofing a - # browser, just avoiding a naive UA fingerprint match. - _USER_AGENT = "ProxMenux-AI-Verifier/1.0" - - def _post_json(self, url: str, payload: dict, headers: dict, - timeout: int = 30) -> dict: - data = json.dumps(payload).encode("utf-8") - headers = {**headers, "User-Agent": self._USER_AGENT} - req = urllib.request.Request(url, data=data, headers=headers, method="POST") - with urllib.request.urlopen(req, timeout=timeout) as resp: - return json.loads(resp.read().decode("utf-8")) - - def _get_json(self, url: str, headers: dict, timeout: int = 30) -> dict: - headers = {**headers, "User-Agent": self._USER_AGENT} - req = urllib.request.Request(url, headers=headers, method="GET") - with urllib.request.urlopen(req, timeout=timeout) as resp: - return json.loads(resp.read().decode("utf-8")) - - -# ══════════════════════════════════════════════════════════════════ -# OpenAI-compatible family (OpenAI, Groq, OpenRouter, LiteLLM, LM -# Studio, vLLM, LocalAI, etc.). They all speak the same endpoints. -# ══════════════════════════════════════════════════════════════════ - -class OpenAICompatProvider(Provider): - name = "openai-compat" - default_base = "https://api.openai.com" - models_path = "/v1/models" - chat_path = "/v1/chat/completions" - - def _base(self) -> str: - return (self.base_url or self.default_base).rstrip("/") - - @staticmethod - def _is_reasoning_model(model: str) -> bool: - """True for OpenAI reasoning models (o-series + non-chat gpt-5+). - - Must be kept in sync with the matching helper in ProxMenux's - openai_provider.py — same rule, same consequence: - - send max_completion_tokens instead of max_tokens - - omit temperature (default is the only accepted value). - """ - m = model.lower() - if len(m) >= 2 and m[0] == "o" and m[1].isdigit(): - return True - if m.startswith("gpt-5") and "-chat" not in m: - return True - return False - - def list_models(self) -> List[str]: - url = f"{self._base()}{self.models_path}" - headers = {"Authorization": f"Bearer {self.api_key}"} - data = self._get_json(url, headers) - return [m.get("id", "") for m in data.get("data", []) if m.get("id")] - - def generate(self, model: str, system: str, user: str, - max_tokens: int = 250, timeout: int = 30) -> str: - url = f"{self._base()}{self.chat_path}" - headers = { - "Authorization": f"Bearer {self.api_key}", - "Content-Type": "application/json", - } - payload = { - "model": model, - "messages": [ - {"role": "system", "content": system}, - {"role": "user", "content": user}, - ], - } - if self._is_reasoning_model(model): - # Reasoning models spend budget on internal reasoning by - # default, which yields empty replies at small max_tokens. - # reasoning_effort=minimal keeps that overhead low so the - # whole budget reaches the user, aligned with the short - # translate+explain task ProxMenux uses. Mirror this in - # ProxMenux's openai_provider.py. - payload["max_completion_tokens"] = max_tokens - payload["reasoning_effort"] = "minimal" - else: - payload["max_tokens"] = max_tokens - payload["temperature"] = 0.3 - data = self._post_json(url, payload, headers, timeout) - try: - return data["choices"][0]["message"]["content"].strip() - except (KeyError, IndexError) as exc: - raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}") - - -class OpenAIProvider(OpenAICompatProvider): - name = "openai" - default_base = "https://api.openai.com" - - -class GroqProvider(OpenAICompatProvider): - name = "groq" - default_base = "https://api.groq.com/openai" - - -class OpenRouterProvider(OpenAICompatProvider): - name = "openrouter" - default_base = "https://openrouter.ai/api" - - -# ══════════════════════════════════════════════════════════════════ -# Gemini — different endpoint shape, keyed via ?key=... query param. -# ══════════════════════════════════════════════════════════════════ - -class GeminiProvider(Provider): - name = "gemini" - default_base = "https://generativelanguage.googleapis.com/v1beta" - - @staticmethod - def _has_thinking_mode(model: str) -> bool: - """True for Gemini variants that enable "thinking" by default. - - Kept in sync with ProxMenux's gemini_provider.py. 2.5+ pro/flash - and 3.x pro/flash consume output tokens on reasoning, which - yields empty replies when max_tokens is small. We pass - thinkingBudget=0 to disable thinking so the short translate+ - explain test sees actual text. Lite variants don't have thinking - enabled and are not flagged here. - """ - m = model.lower() - if "lite" in m: - return False - return m.startswith("gemini-2.5") or m.startswith("gemini-3") - - def list_models(self) -> List[str]: - url = f"{self.default_base}/models?key={self.api_key}" - data = self._get_json(url, {}) - names = [] - for m in data.get("models", []): - raw = m.get("name", "") - if raw.startswith("models/"): - raw = raw[len("models/"):] - # Only keep text-generation capable models. - if "generateContent" in m.get("supportedGenerationMethods", []): - names.append(raw) - return names - - def generate(self, model: str, system: str, user: str, - max_tokens: int = 250, timeout: int = 30) -> str: - url = f"{self.default_base}/models/{model}:generateContent?key={self.api_key}" - gen_config = { - "maxOutputTokens": max_tokens, - "temperature": 0.3, - } - if self._has_thinking_mode(model): - gen_config["thinkingConfig"] = {"thinkingBudget": 0} - payload = { - "system_instruction": {"parts": [{"text": system}]}, - "contents": [{"parts": [{"text": user}]}], - "generationConfig": gen_config, - } - data = self._post_json(url, payload, {"Content-Type": "application/json"}, timeout) - try: - return data["candidates"][0]["content"]["parts"][0]["text"].strip() - except (KeyError, IndexError) as exc: - raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}") - - -# ══════════════════════════════════════════════════════════════════ -# Anthropic — own schema and headers. -# ══════════════════════════════════════════════════════════════════ - -class AnthropicProvider(Provider): - name = "anthropic" - default_base = "https://api.anthropic.com" - version = "2023-06-01" - - def list_models(self) -> List[str]: - url = f"{self.default_base}/v1/models" - headers = { - "x-api-key": self.api_key, - "anthropic-version": self.version, - } - try: - data = self._get_json(url, headers) - return [m.get("id", "") for m in data.get("data", []) if m.get("id")] - except urllib.error.HTTPError: - # Older keys don't have the endpoint; caller decides what to do. - return [] - - def generate(self, model: str, system: str, user: str, - max_tokens: int = 250, timeout: int = 30) -> str: - url = f"{self.default_base}/v1/messages" - headers = { - "x-api-key": self.api_key, - "anthropic-version": self.version, - "Content-Type": "application/json", - } - # ProxMenux's real anthropic_provider.py does NOT send `temperature`, - # and Anthropic's newest generation (Claude Sonnet 5, Opus 4.7 / - # 4.8, Fable 5) now rejects it with: - # invalid_request_error: `temperature` is deprecated for this model. - # Omitting it here aligns the verifier with production behaviour so - # the newest-gen models can be reached during verification. - payload = { - "model": model, - "max_tokens": max_tokens, - "system": system, - "messages": [{"role": "user", "content": user}], - } - data = self._post_json(url, payload, headers, timeout) - try: - return data["content"][0]["text"].strip() - except (KeyError, IndexError) as exc: - raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}") - - -PROVIDERS = { - "openai": OpenAIProvider, - "groq": GroqProvider, - "gemini": GeminiProvider, - "anthropic": AnthropicProvider, - "openrouter": OpenRouterProvider, -} - - -def make_provider(name: str, api_key: str, - base_url: Optional[str] = None) -> Provider: - cls = PROVIDERS.get(name) - if not cls: - raise ValueError(f"unknown provider: {name}") - return cls(api_key, base_url=base_url) diff --git a/.github/scripts/ai-models-verifier/verify.py b/.github/scripts/ai-models-verifier/verify.py deleted file mode 100755 index 4fb92d04..00000000 --- a/.github/scripts/ai-models-verifier/verify.py +++ /dev/null @@ -1,232 +0,0 @@ -#!/usr/bin/env python3 -"""ProxMenux AI-model verifier. - -Runs a standardized translate+explain test against every model each -provider currently advertises, and emits a per-model verdict so the -verified_ai_models.json list can be refreshed with confidence. - -Not packaged with the AppImage — keep this in a private repo alongside -the API keys. - -Usage: - cp .env.example .env # fill in the API keys you have - python3 verify.py # test all providers with keys - python3 verify.py --provider groq # just one - python3 verify.py --provider openai --limit 5 # only first 5 models - python3 verify.py --json-out report.json # machine-readable output too -""" -from __future__ import annotations - -import argparse -import json -import os -import sys -import time -from pathlib import Path -from typing import Dict, List, Optional, Tuple - -from prompts import DOMAIN_HINTS, REQUIRED_SPANISH_HINTS, SYSTEM_PROMPT, USER_MESSAGE -from providers import PROVIDERS, make_provider - - -# Non-chat model name patterns. Skipping these saves test time and keeps -# the report focused on models that could actually serve notifications. -SKIP_PATTERNS = ( - "embedding", "whisper", "tts", "dall-e", "dalle", "image", - "realtime", "audio", "moderation", "search", - "code-search", "text-similarity", "babbage", "davinci", - "curie", "ada", "transcribe", -) - - -def load_env(env_path: Path) -> Dict[str, str]: - """Minimal .env loader (avoids a python-dotenv dependency).""" - if not env_path.exists(): - return {} - env: Dict[str, str] = {} - for line in env_path.read_text().splitlines(): - line = line.strip() - if not line or line.startswith("#") or "=" not in line: - continue - k, v = line.split("=", 1) - env[k.strip()] = v.strip().strip('"').strip("'") - return env - - -def should_skip_model(model: str) -> bool: - m = model.lower() - return any(p in m for p in SKIP_PATTERNS) - - -def assess_response(text: str) -> Tuple[str, List[str]]: - """Classify the model output. Returns (verdict, reasons). - - verdict is one of: - - 'pass': Spanish, on-topic, reasonable length - - 'warn': responded but one heuristic failed (borderline) - - 'fail': empty, wrong language, or off-topic - """ - reasons: List[str] = [] - if not text or len(text) < 30: - return "fail", ["empty or too short response"] - - text_low = " " + text.lower() + " " - spanish_hits = sum(1 for h in REQUIRED_SPANISH_HINTS if h in text_low) - domain_hits = sum(1 for h in DOMAIN_HINTS if h.lower() in text_low) - - if spanish_hits < 3: - reasons.append(f"not Spanish ({spanish_hits}/{len(REQUIRED_SPANISH_HINTS)} hints)") - if domain_hits < 1: - reasons.append("did not engage with the domain") - if len(text) > 1500: - reasons.append("response unusually long") - - if not reasons: - return "pass", [] - # Responded and engaged, but one signal missed → warn (keep in list - # with a caveat; don't auto-include). - if domain_hits >= 1 and spanish_hits >= 1: - return "warn", reasons - return "fail", reasons - - -def run_model(provider, model: str, timeout: int) -> dict: - t0 = time.time() - try: - out = provider.generate( - model, SYSTEM_PROMPT, USER_MESSAGE, - max_tokens=250, timeout=timeout, - ) - latency = time.time() - t0 - verdict, reasons = assess_response(out) - return { - "model": model, - "verdict": verdict, - "latency_s": round(latency, 2), - "reasons": reasons, - "sample": (out[:140] if out else "").replace("\n", " "), - "error": None, - } - except Exception as exc: # HTTPError, ProviderError, timeouts - return { - "model": model, - "verdict": "fail", - "latency_s": round(time.time() - t0, 2), - "reasons": [], - "sample": "", - "error": str(exc)[:200], - } - - -def tag(verdict: str) -> str: - return {"pass": "✓", "warn": "⚠", "fail": "✗"}.get(verdict, "?") - - -def run_provider(name: str, api_key: str, base_url: Optional[str], - timeout: int, limit: Optional[int]) -> dict: - print(f"\n=== {name} ===") - try: - provider = make_provider(name, api_key, base_url=base_url) - models = provider.list_models() - except Exception as exc: - print(f" list_models() failed: {exc}") - return {"provider": name, "error": str(exc), "results": []} - - models = [m for m in models if not should_skip_model(m)] - if limit: - models = models[:limit] - if not models: - print(" (no eligible models)") - return {"provider": name, "error": None, "results": []} - - print(f" discovered {len(models)} model(s)") - results: List[dict] = [] - for m in models: - r = run_model(provider, m, timeout) - results.append(r) - latency = f"{r['latency_s']}s" - if r["error"]: - suffix = f" — {r['error']}" - elif r["reasons"]: - suffix = f" — {'; '.join(r['reasons'])}" - else: - suffix = "" - print(f" {tag(r['verdict'])} {m:<50} {latency:>6}{suffix}") - - return {"provider": name, "error": None, "results": results} - - -def summarize(all_results: List[dict]) -> None: - print("\n" + "=" * 64) - print("Suggested verified_ai_models.json entries (passing models only)") - print("=" * 64) - any_output = False - for pr in all_results: - if pr.get("error"): - continue - passed = [r for r in pr["results"] if r["verdict"] == "pass"] - if not passed: - continue - any_output = True - passed_sorted = sorted(passed, key=lambda x: x["latency_s"]) - print(f'\n "{pr["provider"]}": {{') - print(' "models": [') - for r in passed_sorted: - print(f' "{r["model"]}",') - print(" ],") - print(f' "recommended": "{passed_sorted[0]["model"]}"') - print(" },") - - warn_total = sum( - 1 for pr in all_results for r in pr["results"] - if r["verdict"] == "warn" - ) - if warn_total: - print(f"\n Note: {warn_total} model(s) came back as ⚠ (warn) — review those manually.") - if not any_output: - print("\n (no models passed; check keys and network)") - - -def main() -> int: - ap = argparse.ArgumentParser( - description="Verify AI models for ProxMenux notification enrichment." - ) - ap.add_argument("--env", default=".env", help=".env file path") - ap.add_argument("--provider", action="append", default=[], - help="run a specific provider (repeat to add more)") - ap.add_argument("--timeout", type=int, default=30, - help="seconds per request (default: 30)") - ap.add_argument("--limit", type=int, default=None, - help="test at most N models per provider (debug)") - ap.add_argument("--json-out", default=None, - help="write machine-readable report to this path") - args = ap.parse_args() - - env = {**os.environ, **load_env(Path(args.env))} - - provider_list = args.provider or list(PROVIDERS.keys()) - tested: List[dict] = [] - for name in provider_list: - if name not in PROVIDERS: - print(f"unknown provider: {name}", file=sys.stderr) - continue - key_var = f"{name.upper()}_API_KEY" - url_var = f"{name.upper()}_BASE_URL" - api_key = env.get(key_var, "") - base_url = env.get(url_var) or None - if not api_key: - print(f"\n=== {name} ===\n skipped — {key_var} not set") - continue - tested.append(run_provider(name, api_key, base_url, args.timeout, args.limit)) - - summarize(tested) - - if args.json_out: - Path(args.json_out).write_text(json.dumps(tested, indent=2)) - print(f"\nReport written to {args.json_out}") - - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/.github/workflows/verify-ai-models.yml b/.github/workflows/verify-ai-models.yml deleted file mode 100644 index 6fe891ed..00000000 --- a/.github/workflows/verify-ai-models.yml +++ /dev/null @@ -1,103 +0,0 @@ -name: Verify AI models catalog - -# Runs the AI-model verifier and commits any changes to -# AppImage/config/verified_ai_models.json on the same branch the run -# was launched from. -# -# GitHub only fires `on: schedule` from the default branch, so the -# daily cron always runs against main and keeps stable users fresh. -# When a beta cycle needs its own refresh on develop, use the -# "Run workflow" button on the Actions tab and pick develop from the -# branch selector — the same YAML then checks out develop, runs the -# verifier and commits back to develop. Cross-branch pushes never -# happen: each run only touches the branch it started on. -# -# The verifier code lives at .github/scripts/ai-models-verifier/ and -# reads API keys from repository Secrets. Any provider without a key -# is skipped silently — the workflow keeps going with the rest. - -on: - schedule: - - cron: '0 4 * * *' # 04:00 UTC every day — cron always fires from main - workflow_dispatch: # manual trigger — branch is picked in the UI - -permissions: - contents: write - -concurrency: - # Keyed by branch so a manual develop run does not collide with the - # scheduled main run — each branch gets its own serialisation lane. - group: verify-ai-models-${{ github.ref_name }} - cancel-in-progress: false - -jobs: - verify: - runs-on: ubuntu-latest - timeout-minutes: 20 - steps: - - name: Check out the branch this run belongs to - uses: actions/checkout@v4 - with: - # `github.ref_name` resolves to main for the cron and to the - # branch selected in the dispatch UI otherwise. The same - # value is used again below when we push, so every run is - # symmetric: checkout X → refresh → push X. - ref: ${{ github.ref_name }} - token: ${{ secrets.GITHUB_TOKEN }} - - - name: Set up Python - uses: actions/setup-python@v5 - with: - python-version: '3.12' - - - name: Run verifier - working-directory: .github/scripts/ai-models-verifier - env: - # API keys — each is optional. verify.py silently skips any - # provider whose *_API_KEY env var is empty, so the workflow - # runs even when only a subset of keys is configured. - OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} - ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} - GROQ_API_KEY: ${{ secrets.GROQ_API_KEY }} - GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }} - OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }} - # Optional base URLs for custom-endpoint providers. - OPENAI_BASE_URL: ${{ secrets.OPENAI_BASE_URL }} - run: | - python3 verify.py --json-out /tmp/report.json || true - if [ ! -s /tmp/report.json ]; then - echo "Verifier produced no report — bailing" - exit 1 - fi - - - name: Apply report to catalog - id: apply - working-directory: . - run: | - set +e - python3 .github/scripts/ai-models-verifier/apply.py \ - --report /tmp/report.json \ - --catalog AppImage/config/verified_ai_models.json - code=$? - set -e - case "$code" in - 0) echo "changed=false" >> "$GITHUB_OUTPUT" ;; - 10) echo "changed=true" >> "$GITHUB_OUTPUT" ;; - *) echo "apply.py exited with $code"; exit "$code" ;; - esac - - - name: Commit and push back to the same branch - if: steps.apply.outputs.changed == 'true' - run: | - git config user.name "proxmenux-bot" - git config user.email "proxmenux-bot@users.noreply.github.com" - git add AppImage/config/verified_ai_models.json - git commit -m "chore(ai-models): daily catalog refresh" - # Push to the branch this run started on — same ref used at - # checkout above, so the operation is symmetric regardless of - # whether cron (main) or dispatch (any branch) triggered it. - git push origin HEAD:${{ github.ref_name }} - - - name: Report unchanged - if: steps.apply.outputs.changed != 'true' - run: echo "Catalog already up to date — nothing to commit." diff --git a/AppImage/components/notification-settings.tsx b/AppImage/components/notification-settings.tsx index 03f7108b..fe7b54df 100644 --- a/AppImage/components/notification-settings.tsx +++ b/AppImage/components/notification-settings.tsx @@ -341,16 +341,6 @@ export function NotificationSettings() { const [aiTestResult, setAiTestResult] = useState<{ success: boolean; message: string; model?: string } | null>(null) const [providerModels, setProviderModels] = useState([]) const [loadingProviderModels, setLoadingProviderModels] = useState(false) - // Metadata returned by the backend after the catalog refresh — used - // to render the "Last update" line and toast the diff. - const [catalogMeta, setCatalogMeta] = useState<{ - success: boolean - message: string - changed: boolean - previous_updated: string | null - new_updated: string | null - diff: Record - } | null>(null) const [showCustomPromptInfo, setShowCustomPromptInfo] = useState(false) const [editingCustomPrompt, setEditingCustomPrompt] = useState(false) const [customPromptDraft, setCustomPromptDraft] = useState("") @@ -1003,38 +993,16 @@ export function NotificationSettings() { setLoadingProviderModels(true) try { - const data = await fetchApi<{ - success: boolean - models: string[] - recommended: string - message: string - catalog_meta?: { - success: boolean - message: string - changed: boolean - previous_updated: string | null - new_updated: string | null - diff: Record - } | null - }>("/api/notifications/provider-models", { + const data = await fetchApi<{ success: boolean; models: string[]; recommended: string; message: string }>("/api/notifications/provider-models", { method: "POST", headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ + body: JSON.stringify({ provider, api_key: apiKey, ollama_url: config.ai_ollama_url, openai_base_url: config.ai_openai_base_url, - // Every click pulls the latest verified catalog from GitHub - // before the intersection runs, so a user with an outdated - // AppImage still sees today's model list. The backend keeps - // the local file when the fetch fails, so an offline node - // simply doesn't refresh but still resolves models. - refresh_catalog: true, }), }) - if (data.catalog_meta) { - setCatalogMeta(data.catalog_meta) - } if (data.success && data.models && data.models.length > 0) { setProviderModels(data.models) // Auto-select recommended model if current selection is empty or not in the list @@ -2441,7 +2409,7 @@ export function NotificationSettings() { className="h-9 px-3 shrink-0" onClick={() => fetchProviderModels()} disabled={ - loadingProviderModels || + loadingProviderModels || (config.ai_provider === 'ollama' && !config.ai_ollama_url) || (config.ai_provider !== 'ollama' && config.ai_provider !== 'anthropic' && !config.ai_api_keys?.[config.ai_provider]) } @@ -2451,12 +2419,7 @@ export function NotificationSettings() { ) : ( <> - {/* Label swaps once we have models loaded — "Load" reads - right on the empty state, "Update" reads right after - the first pull. Same action underneath either way. */} - {t(providerModels.length === 0 - ? "settings.notifications.ai.load" - : "settings.notifications.ai.updateModels")} + {t("settings.notifications.ai.load")} )} @@ -2464,32 +2427,6 @@ export function NotificationSettings() { {providerModels.length > 0 && (

{t("settings.notifications.ai.modelsAvailable", { count: providerModels.length })}

)} - {/* Catalog freshness line — surfaces the `_updated` date of the - verified_ai_models.json currently in use, plus a compact diff - summary from the most recent GitHub pull so the user knows - what just happened without opening a modal. */} - {catalogMeta && ( -
-
- {t("settings.notifications.ai.catalogLastUpdated", { - date: catalogMeta.new_updated || catalogMeta.previous_updated || "—", - })} -
- {catalogMeta.changed && catalogMeta.success && Object.keys(catalogMeta.diff).length > 0 && ( -
- {t("settings.notifications.ai.catalogChanged", { - added: Object.values(catalogMeta.diff).reduce((n, d) => n + d.added.length, 0), - removed: Object.values(catalogMeta.diff).reduce((n, d) => n + d.removed.length, 0), - })} -
- )} - {!catalogMeta.success && catalogMeta.message && ( -
- {t("settings.notifications.ai.catalogFetchFailed", { reason: catalogMeta.message })} -
- )} -
- )} {/* Prompt Mode section */} diff --git a/AppImage/config/verified_ai_models.json b/AppImage/config/verified_ai_models.json index 23c0dbd2..85077cbc 100644 --- a/AppImage/config/verified_ai_models.json +++ b/AppImage/config/verified_ai_models.json @@ -1,123 +1,85 @@ { "_description": "Verified AI models for ProxMenux notifications. Only models listed here will be shown to users. Models are tested to work with the chat/completions API format.", - "_updated": "2026-09-02", - "_verifier": "Refreshed by .github/workflows/verify-ai-models.yml (daily). The workflow runs .github/scripts/ai-models-verifier/verify.py against every provider whose API key is configured in repository Secrets, then applies the report via apply.py — which honours per-provider `_exclude` lists. `_exclude` is intentionally minimal: it only drops models that are technically incapable of generating a chat completion for a normal prompt (safety classifiers, agentic-only endpoints, meta-routers, wrong modalities). Everything else the verifier passes is surfaced — including language-specialised models (Arabic, Chinese, ...), reasoning models, legacy families and dated snapshots — so a user with a specific need can still pick the model that fits. Manually re-run from the Actions tab (any branch) when a new model needs to be picked up out of cycle.", + "_updated": "2026-07-14", + "_verifier": "Refreshed with tools/ai-models-verifier (private). Re-run before each ProxMenux release to keep the list current. The verifier and ProxMenux share the same reasoning/thinking-model handlers so their verdicts stay aligned with runtime behaviour.", + "groq": { "models": [ - "allam-2-7b", - "qwen/qwen3.8-27b", - "openai/gpt-oss-120b" + "llama-3.3-70b-versatile", + "llama-3.1-8b-instant", + "meta-llama/llama-4-scout-17b-16e-instruct", + "openai/gpt-oss-120b", + "openai/gpt-oss-20b" ], - "recommended": "allam-2-7b", - "_exclude": [ - "openai/gpt-oss-safeguard-*", - "groq/compound", - "groq/compound-*" - ], - "_note": "`_exclude` covers models the verifier may technically pass but that do not produce a usable chat completion: openai/gpt-oss-safeguard-* is a safety classifier (returns a category, not free text); groq/compound* is an agentic system that expects multi-step tool use, not a plain prompt." + "recommended": "llama-3.3-70b-versatile", + "_note": "Verified functionally 2026-07-14 with the Groq API (15 models discovered, 9 passed). Legacy llama-3.1-70b-versatile / llama3-70b-8192 / llama3-8b-8192 / mixtral-8x7b-32768 / gemma2-9b-it removed (retired upstream). llama-4-scout added (current-gen Llama 4, 0.47s). openai/gpt-oss-120b / gpt-oss-20b confirmed. Passing but excluded: allam-2-7b (Arabic-focused), qwen/qwen3-32b (Chinese-first, unreliable Spanish output), openai/gpt-oss-safeguard-20b (safety-classifier variant), groq/compound-mini (agentic system, wrong fit for notification translation)." }, + "gemini": { "models": [ - "gemini-2.5-flash-lite", - "gemini-3.1-flash-lite-preview", - "gemini-3.1-flash-lite", "gemini-flash-lite-latest", - "gemini-3.5-flash-lite", + "gemini-2.5-flash-lite", "gemini-2.5-flash", - "gemini-3.5-flash", + "gemini-3.1-flash-lite", "gemini-3-flash-preview", - "gemma-4-26b-a4b-it" + "gemini-3.5-flash" ], "recommended": "gemini-2.5-flash-lite", - "_exclude": [ - "gemini-embedding-*", - "gemini-*-pro*", - "gemini-*-thinking*" - ], - "_deprecated": [ - "gemini-2.0-flash", - "gemini-2.0-flash-lite", - "gemini-1.5-flash", - "gemini-1.0-pro", - "gemini-pro" - ], - "_note": "`_exclude` drops embeddings (wrong modality) and Pro / thinking variants that reject `thinkingConfig.thinkingBudget: 0` and therefore never return a visible completion within a reasonable token budget — technical failure with our current provider config." + "_note": "Verified 2026-07-13. gemini-flash-lite-latest now passes consistently (1.6s) and is fastest, but gemini-2.5-flash-lite remains recommended because 'latest' aliases can drift over time. gemini-3.1-flash-lite is the stable successor to 3-flash-preview. Pro variants continue to reject thinkingBudget=0 and are overkill for notification translation.", + "_deprecated": ["gemini-2.0-flash", "gemini-2.0-flash-lite", "gemini-1.5-flash", "gemini-1.0-pro", "gemini-pro"] }, + "openai": { "models": [ - "gpt-4.1-nano-2025-04-14", "gpt-4.1-nano", - "gpt-4.1-2025-04-14", - "gpt-4.1", - "gpt-4o-2024-05-13", - "gpt-4o-2024-11-20", - "gpt-5-nano", - "gpt-5-nano-2025-08-07", - "gpt-3.5-turbo-1106", - "gpt-3.5-turbo-0125", - "gpt-4o-mini-2024-07-18", - "gpt-4.1-mini-2025-04-14", - "gpt-4o-mini", - "gpt-3.5-turbo-16k", "gpt-4.1-mini", - "gpt-3.5-turbo", + "gpt-4o-mini", + "gpt-4.1", "gpt-4o", - "gpt-4", - "gpt-4-0613", - "gpt-4-turbo", - "gpt-4-turbo-2024-04-09", - "gpt-4o-2024-08-06" + "gpt-5-chat-latest", + "gpt-5-nano" ], "recommended": "gpt-4.1-nano", - "_exclude": [ - "gpt-4o-audio*", - "gpt-4o-realtime*", - "gpt-4o-search*", - "gpt-4o-transcribe*", - "gpt-4o-mini-audio*", - "gpt-4o-mini-realtime*", - "gpt-4o-mini-search*", - "gpt-4o-mini-transcribe*", - "gpt-4o-mini-tts", - "computer-use-*" - ], - "_note": "`_exclude` covers wrong-modality variants (audio, realtime, search, transcribe, tts) that cannot handle a plain notification-translation prompt, plus computer-use which requires an agent loop. All other OpenAI chat/completion models are surfaced — including legacy families (gpt-3.5, gpt-4), reasoning models (o-series, gpt-5.x non-chat) and dated snapshots — so users can pick by their own criteria (cost, quality, reproducibility). openai_provider.py already handles reasoning models via max_completion_tokens + reasoning_effort=minimal." + "_note": "Verified 2026-07-13. gpt-5.4-nano / gpt-5.4-mini removed (HTTP 400 — provider params rejected). gpt-5-nano added (2.0s, current-gen fast). Reasoning models (o-series, gpt-5/5.1/5.2 non-chat variants) are supported by openai_provider.py via max_completion_tokens + reasoning_effort=minimal, but not listed here: their latency is higher and they do not improve translation quality for notifications. Add specific reasoning IDs to this list only if a user explicitly wants them." }, + "anthropic": { "models": [ - "claude-haiku-4-5-20251001", + "claude-haiku-4-5", + "claude-sonnet-5", "claude-opus-4-8", - "claude-opus-4-5-20251101", - "claude-opus-4-7", - "claude-fable-5", - "claude-sonnet-4-5-20250929", - "claude-fable-5-1", + "claude-sonnet-4-6", "claude-opus-4-6", - "claude-sonnet-4-6" + "claude-fable-5" ], - "recommended": "claude-haiku-4-5-20251001", - "_exclude": [], - "_note": "No technical exclusions — every Claude generation returns free-text completions for a plain prompt. The verifier tests each model listed under `models`; if a specific ID stops working upstream it simply drops out of the passing set. Anthropic does not expose a public models-list API, so new models must be added to `models` manually before the verifier can test them." + "recommended": "claude-haiku-4-5", + "_note": "Verified 2026-07-13 with all 10 discovered models passing after aligning the verifier with anthropic_provider.py (temperature omitted — newest generations reject it with 'temperature is deprecated for this model'). Legacy claude-3-5-haiku-latest / claude-3-5-sonnet-latest / claude-3-opus-latest removed (deprecated upstream, not in the Models API). haiku-4-5 is the sweet spot for notification translation (3.6s, $1/$5 per MTok); sonnet-5 for slightly richer output (3.1s, $3/$15); opus-4-8 / fable-5 for demanding cases." }, + "openrouter": { "models": [ - "minimax/minimax-m3:free", - "poolside/laguna-s-2.1:free" + "meta-llama/llama-3.3-70b-instruct", + "meta-llama/llama-3.1-70b-instruct", + "meta-llama/llama-3.1-8b-instruct", + "meta-llama/llama-4-scout", + "anthropic/claude-haiku-4.5", + "anthropic/claude-sonnet-4.6", + "google/gemini-2.5-flash-lite", + "google/gemini-2.5-flash", + "openai/gpt-4o-mini", + "mistralai/mistral-small-3.2-24b-instruct", + "nvidia/nemotron-3-super-120b-a12b:free", + "google/gemma-4-26b-a4b-it:free", + "nvidia/nemotron-nano-12b-v2-vl:free", + "nvidia/nemotron-3-nano-30b-a3b:free", + "poolside/laguna-s-2.1:free", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free", + "openai/gpt-oss-20b:free" ], - "recommended": "minimax/minimax-m3:free", - "_exclude": [ - "openrouter/*", - "*/*-audio*", - "*/*-audio-*", - "*/*-tts*", - "*/*-whisper*", - "*/*-embed*", - "*/*-embedding*", - "*/*-image*", - "*/*-vision-only*" - ], - "_note": "OpenRouter aggregates hundreds of models; the free-tier variants (:free suffix) are intentionally supported per user request and never blocked. `_exclude` covers only meta-routers (`openrouter/free`, `openrouter/auto` — they route dynamically to something else, so their behaviour is not the model the user picked) and wrong-modality models (audio, tts, whisper, embeddings, image, vision-only). Everything else — chat models across every family, language and price tier — is surfaced so the user can pick the fit." + "recommended": "meta-llama/llama-3.3-70b-instruct", + "_note": "Paid tier verified functionally 2026-07-14 with the OpenRouter API — all 10 curated candidates pass the Spanish-translation notification test. Fastest: llama-4-scout (0.51s), gemini-2.5-flash-lite (1.14s), gemini-2.5-flash (1.94s), llama-3.3-70b-instruct (2.29s), claude-haiku-4.5 (2.71s). Free tier verified 2026-08-17 — 7 :free models pass and are appended, ordered by latency: nemotron-3-super-120b-a12b (3.5s), gemma-4-26b-a4b-it (4.1s), nemotron-nano-12b-v2-vl (5.3s), nemotron-3-nano-30b-a3b (5.8s), laguna-s-2.1 (8.3s), nemotron-3-nano-omni-30b-a3b-reasoning (10.7s), gpt-oss-20b (12.2s). Free-tier rate limits (~20 req/min shared across all OpenRouter free users on that model) may cause 429 in high-traffic windows — usable for occasional notification translation, not for high-volume automation. recommended kept as llama-3.3-70b for capability/latency balance; llama-4-scout is a faster alternative worth considering as recommended after a broader release." }, + "ollama": { "_note": "Ollama models are local, we don't filter them. User manages their own models.", "models": [], diff --git a/AppImage/messages/de/common.json b/AppImage/messages/de/common.json index 8c10881f..7264b8e8 100644 --- a/AppImage/messages/de/common.json +++ b/AppImage/messages/de/common.json @@ -1922,11 +1922,7 @@ "gemini": "Es ist eine kostenlose Stufe mit einem guten Preis-Leistungs-Verhältnis verfügbar.", "ollama": "Verwendet Modelle auf Ihrem Ollama-Server. Völlig lokal, privat und kostenlos nutzbar.", "openrouter": "Zugriff auf mehr als 100 Modelle über einen API-Schlüssel." - }, - "updateModels": "Aktualisieren", - "catalogLastUpdated": "Katalog zuletzt aktualisiert: {date}", - "catalogChanged": "Katalog aktualisiert — {added} hinzugefügt, {removed} entfernt", - "catalogFetchFailed": "Katalog konnte nicht aktualisiert werden: {reason}. Lokale Kopie wird verwendet." + } }, "telegramGuide": { "title": "Anleitung zur Einrichtung des Telegram-Bots", diff --git a/AppImage/messages/en/common.json b/AppImage/messages/en/common.json index cdab38b7..d4b39f2f 100644 --- a/AppImage/messages/en/common.json +++ b/AppImage/messages/en/common.json @@ -1921,11 +1921,7 @@ "gemini": "A free tier is available, with a good quality-to-price ratio.", "ollama": "Uses models on your Ollama server. Fully local, private and free to run.", "openrouter": "Access to more than 100 models through one API key." - }, - "updateModels": "Update", - "catalogLastUpdated": "Verified catalog last updated: {date}", - "catalogChanged": "Catalog refreshed — {added} added, {removed} removed", - "catalogFetchFailed": "Could not refresh catalog: {reason}. Using local copy." + } }, "telegramGuide": { "title": "Telegram bot setup guide", diff --git a/AppImage/messages/es/common.json b/AppImage/messages/es/common.json index df7bcd27..3f738abd 100644 --- a/AppImage/messages/es/common.json +++ b/AppImage/messages/es/common.json @@ -1922,11 +1922,7 @@ "gemini": "Hay disponible un nivel gratuito, con una buena relación calidad-precio.", "ollama": "Utiliza modelos en su servidor Ollama. Totalmente local, privado y gratuito.", "openrouter": "Acceso a más de 100 modelos a través de una clave API." - }, - "updateModels": "Actualizar", - "catalogLastUpdated": "Última actualización del catálogo verificado: {date}", - "catalogChanged": "Catálogo actualizado — {added} añadidos, {removed} retirados", - "catalogFetchFailed": "No se pudo actualizar el catálogo: {reason}. Se usa la copia local." + } }, "telegramGuide": { "title": "Guía de configuración del bot de Telegram", diff --git a/AppImage/messages/fr/common.json b/AppImage/messages/fr/common.json index d648578e..2857230b 100644 --- a/AppImage/messages/fr/common.json +++ b/AppImage/messages/fr/common.json @@ -1922,11 +1922,7 @@ "gemini": "Un niveau gratuit est disponible, avec un bon rapport qualité-prix.", "ollama": "Utilise des modèles sur votre serveur Ollama. Entièrement local, privé et gratuit.", "openrouter": "Accès à plus de 100 modèles via une seule clé API." - }, - "updateModels": "Actualiser", - "catalogLastUpdated": "Dernière mise à jour du catalogue vérifié : {date}", - "catalogChanged": "Catalogue mis à jour — {added} ajoutés, {removed} retirés", - "catalogFetchFailed": "Impossible d'actualiser le catalogue : {reason}. Copie locale utilisée." + } }, "telegramGuide": { "title": "Guide de configuration du robot Telegram", diff --git a/AppImage/messages/it/common.json b/AppImage/messages/it/common.json index f1bdccb4..8a4f7ef5 100644 --- a/AppImage/messages/it/common.json +++ b/AppImage/messages/it/common.json @@ -1922,11 +1922,7 @@ "gemini": "È disponibile un livello gratuito, con un buon rapporto qualità-prezzo.", "ollama": "Utilizza i modelli sul tuo server Ollama. Completamente locale, privato e gratuito.", "openrouter": "Accesso a più di 100 modelli tramite una chiave API." - }, - "updateModels": "Aggiorna", - "catalogLastUpdated": "Ultimo aggiornamento del catalogo verificato: {date}", - "catalogChanged": "Catalogo aggiornato — {added} aggiunti, {removed} rimossi", - "catalogFetchFailed": "Impossibile aggiornare il catalogo: {reason}. Verrà usata la copia locale." + } }, "telegramGuide": { "title": "Guida alla configurazione del bot di Telegram", diff --git a/AppImage/messages/pt/common.json b/AppImage/messages/pt/common.json index 3eddffdf..b031fad8 100644 --- a/AppImage/messages/pt/common.json +++ b/AppImage/messages/pt/common.json @@ -1922,11 +1922,7 @@ "gemini": "Um nível gratuito está disponível, com uma boa relação qualidade/preço.", "ollama": "Usa modelos em seu servidor Ollama. Totalmente local, privado e de operação gratuita.", "openrouter": "Acesso a mais de 100 modelos através de uma chave API." - }, - "updateModels": "Atualizar", - "catalogLastUpdated": "Última atualização do catálogo verificado: {date}", - "catalogChanged": "Catálogo atualizado — {added} adicionados, {removed} removidos", - "catalogFetchFailed": "Não foi possível atualizar o catálogo: {reason}. A usar a cópia local." + } }, "telegramGuide": { "title": "Guia de configuração do bot do Telegram", diff --git a/AppImage/messages/sk/common.json b/AppImage/messages/sk/common.json index e11c706f..d562897c 100644 --- a/AppImage/messages/sk/common.json +++ b/AppImage/messages/sk/common.json @@ -1921,11 +1921,7 @@ "gemini": "Ponúka bezplatnú úroveň a dobrý pomer kvality a ceny.", "ollama": "Používa modely na vašom Ollama serveri. Beží lokálne, súkromne a bez poplatkov.", "openrouter": "Prístup k viac než 100 modelom cez jeden API kľúč." - }, - "updateModels": "Aktualizovať", - "catalogLastUpdated": "Posledná aktualizácia overeného katalógu: {date}", - "catalogChanged": "Katalóg aktualizovaný — {added} pridaných, {removed} odstránených", - "catalogFetchFailed": "Katalóg nebolo možné aktualizovať: {reason}. Používa sa lokálna kópia." + } }, "telegramGuide": { "title": "Nastavenie Telegram bota", diff --git a/AppImage/messages/sv/common.json b/AppImage/messages/sv/common.json index 856f62e5..3b7beada 100644 --- a/AppImage/messages/sv/common.json +++ b/AppImage/messages/sv/common.json @@ -1922,11 +1922,7 @@ "gemini": "En gratis nivå är tillgänglig, med ett bra förhållande mellan kvalitet och pris.", "ollama": "Använder modeller på din Ollama-server. Helt lokalt, privat och gratis att köra.", "openrouter": "Tillgång till mer än 100 modeller via en API-nyckel." - }, - "updateModels": "Uppdatera", - "catalogLastUpdated": "Verifierad katalog senast uppdaterad: {date}", - "catalogChanged": "Katalog uppdaterad — {added} tillagda, {removed} borttagna", - "catalogFetchFailed": "Katalogen kunde inte uppdateras: {reason}. Använder lokal kopia." + } }, "telegramGuide": { "title": "Installationsguide för Telegram bot", diff --git a/AppImage/scripts/flask_notification_routes.py b/AppImage/scripts/flask_notification_routes.py index 548b8e26..13ddc42f 100644 --- a/AppImage/scripts/flask_notification_routes.py +++ b/AppImage/scripts/flask_notification_routes.py @@ -402,190 +402,32 @@ def test_notification(): return jsonify({'error': f'Internal error ({type(e).__name__})'}), 500 -_VERIFIED_MODELS_REMOTE_URL = ( - "https://raw.githubusercontent.com/MacRimi/ProxMenux/main/" - "AppImage/config/verified_ai_models.json" -) - - -def _resolve_verified_models_path() -> Path: - """Locate the on-disk verified_ai_models.json used at runtime. - - AppImage layout keeps scripts and config under /usr/bin/; the dev - tree has them one level apart. We probe both and return whichever - exists so `load_verified_models` and the refresh helper agree on the - same file — otherwise a refresh writing to one path and a read - hitting the other would silently do nothing. - """ - script_dir = Path(__file__).parent - candidate = script_dir / 'config' / 'verified_ai_models.json' - if candidate.exists(): - return candidate - dev_candidate = script_dir.parent / 'config' / 'verified_ai_models.json' - if dev_candidate.exists(): - return dev_candidate - # No file exists yet — return the AppImage-shaped path so a fresh - # refresh has somewhere to write. The parent dir is created if needed. - candidate.parent.mkdir(parents=True, exist_ok=True) - return candidate - - def load_verified_models(): """Load verified models from config file. - - Returns {} if the file is missing or unreadable; callers already - handle the empty case by returning provider defaults or the API's - unfiltered list. + + Checks multiple paths: + 1. Same directory as script (AppImage: /usr/bin/config/) + 2. Parent directory config folder (dev: AppImage/config/) """ try: - path = _resolve_verified_models_path() - if path.exists(): - with open(path, 'r') as f: + # Try AppImage path first (scripts and config both in /usr/bin/) + script_dir = Path(__file__).parent + config_path = script_dir / 'config' / 'verified_ai_models.json' + + if not config_path.exists(): + # Try development path (AppImage/scripts/ -> AppImage/config/) + config_path = script_dir.parent / 'config' / 'verified_ai_models.json' + + if config_path.exists(): + with open(config_path, 'r') as f: return json.load(f) - print(f"[flask_notification_routes] Config not found at {path}") + else: + print(f"[flask_notification_routes] Config not found at {config_path}") except Exception as e: print(f"[flask_notification_routes] Failed to load verified models: {e}") return {} -def _catalog_signature(catalog: dict) -> dict: - """Reduce a catalog to `{provider: [sorted models]}`. - - Used to diff the local vs the remote catalog without dragging the - unrelated `_note` / `_deprecated` / `_updated` metadata into the - comparison. - """ - sig = {} - for provider, block in catalog.items(): - if isinstance(block, dict) and isinstance(block.get('models'), list): - sig[provider] = sorted(str(m) for m in block['models']) - return sig - - -def refresh_verified_models_from_github() -> dict: - """Fetch the canonical catalog from the main branch and overwrite the - local copy in-place. - - Returns a dict shaped for the API response: - { - 'success': True|False, - 'message': '...', - 'changed': bool, - 'previous_updated': 'YYYY-MM-DD' | None, - 'new_updated': 'YYYY-MM-DD' | None, - 'diff': { : {'added': [...], 'removed': [...]} }, - } - - Never touches the local file when the remote fetch fails — the - installed catalog remains authoritative in the offline / degraded - case, which is the deliberate behaviour the UI relies on to show - 'the catalog is still active' after a failed refresh. - """ - import urllib.request - import urllib.error - - result = { - 'success': False, - 'message': '', - 'changed': False, - 'previous_updated': None, - 'new_updated': None, - 'diff': {}, - } - - try: - req = urllib.request.Request(_VERIFIED_MODELS_REMOTE_URL, method='GET') - req.add_header('User-Agent', 'ProxMenux-Monitor/1.1') - req.add_header('Accept', 'application/json') - with urllib.request.urlopen(req, timeout=15) as resp: - raw = resp.read().decode('utf-8') - remote = json.loads(raw) - except urllib.error.HTTPError as e: - result['message'] = f'GitHub returned HTTP {e.code}' - return result - except urllib.error.URLError as e: - result['message'] = f'Could not reach GitHub: {e.reason}' - return result - except (json.JSONDecodeError, ValueError) as e: - result['message'] = f'Remote catalog is not valid JSON: {e}' - return result - except Exception as e: - result['message'] = f'Refresh failed: {type(e).__name__}: {e}' - return result - - if not isinstance(remote, dict) or not any( - isinstance(remote.get(k), dict) and 'models' in remote[k] - for k in remote.keys() - if not k.startswith('_') - ): - result['message'] = 'Remote catalog has an unexpected shape' - return result - - local = load_verified_models() or {} - prev_sig = _catalog_signature(local) - new_sig = _catalog_signature(remote) - - diff = {} - for provider in sorted(set(prev_sig) | set(new_sig)): - prev_models = set(prev_sig.get(provider, [])) - new_models = set(new_sig.get(provider, [])) - added = sorted(new_models - prev_models) - removed = sorted(prev_models - new_models) - if added or removed: - diff[provider] = {'added': added, 'removed': removed} - - result['previous_updated'] = local.get('_updated') - result['new_updated'] = remote.get('_updated') - result['changed'] = bool(diff) or (result['previous_updated'] != result['new_updated']) - result['diff'] = diff - - # Only touch disk when something actually changed to keep mtimes - # meaningful and avoid a spurious modification in backup diffs. - if result['changed']: - try: - path = _resolve_verified_models_path() - tmp = path.with_suffix(path.suffix + '.tmp') - with open(tmp, 'w') as f: - json.dump(remote, f, indent=2, ensure_ascii=False) - f.write('\n') - tmp.replace(path) - result['success'] = True - result['message'] = 'Catalog updated' - except Exception as e: - result['success'] = False - result['message'] = f'Downloaded but failed to save: {type(e).__name__}: {e}' - return result - else: - result['success'] = True - result['message'] = 'Catalog already up to date' - return result - - -@notification_bp.route('/api/notifications/refresh-model-catalog', methods=['POST']) -@require_auth -def refresh_model_catalog(): - """Pull the verified_ai_models.json from ProxMenux/main on GitHub and - overwrite the local copy. Returns the diff so the UI can toast a - meaningful summary (added / removed models per provider) instead of a - generic 'refreshed' message. - - Standalone endpoint so an admin can refresh without also fetching - provider models; the provider-models call carries a shortcut flag - for the common 'refresh + load' click from the Notifications UI. - """ - try: - return jsonify(refresh_verified_models_from_github()) - except Exception as e: - return jsonify({ - 'success': False, - 'message': f'Refresh failed: {type(e).__name__}: {e}', - 'changed': False, - 'previous_updated': None, - 'new_updated': None, - 'diff': {}, - }), 500 - - @notification_bp.route('/api/notifications/provider-models', methods=['POST']) @require_auth def get_provider_models(): @@ -617,25 +459,9 @@ def get_provider_models(): api_key = _resolve_masked_api_key(provider, data.get('api_key', '')) ollama_url = data.get('ollama_url', 'http://localhost:11434') openai_base_url = data.get('openai_base_url', '') - # `refresh_catalog=true` in the request body triggers a GitHub - # pull of verified_ai_models.json before the intersection runs. - # The Notifications UI passes this on every Load / Update click - # so the catalog stays fresh without a separate button. - refresh_catalog = bool(data.get('refresh_catalog', False)) - catalog_meta = None - if refresh_catalog: - catalog_meta = refresh_verified_models_from_github() - - def _reply(payload, status=200): - # Every response from this endpoint carries `catalog_meta` - # when a refresh was requested, so the UI can render the - # diff toast + "last updated" line from a single roundtrip. - if catalog_meta is not None: - payload = {**payload, 'catalog_meta': catalog_meta} - return jsonify(payload), status if not provider: - return _reply({'success': False, 'models': [], 'message': 'Provider not specified'}) + return jsonify({'success': False, 'models': [], 'message': 'Provider not specified'}) # SSRF guard before we touch the URL. Ollama is local-by-design so # loopback is allowed there; OpenAI base URL must be a real external @@ -643,11 +469,11 @@ def get_provider_models(): if provider == 'ollama': ok, err = validate_external_url(ollama_url, allow_loopback=True) if not ok: - return _reply({'success': False, 'models': [], 'message': f'Invalid ollama_url: {err}'}, 400) + return jsonify({'success': False, 'models': [], 'message': f'Invalid ollama_url: {err}'}), 400 if provider == 'openai' and openai_base_url: ok, err = validate_external_url(openai_base_url, allow_loopback=False) if not ok: - return _reply({'success': False, 'models': [], 'message': f'Invalid openai_base_url: {err}'}, 400) + return jsonify({'success': False, 'models': [], 'message': f'Invalid openai_base_url: {err}'}), 400 # Load verified models config verified_config = load_verified_models() @@ -668,13 +494,13 @@ def get_provider_models(): result = json.loads(resp.read().decode('utf-8')) models = [m.get('name', '') for m in result.get('models', []) if m.get('name')] models = sorted(models) - return _reply({ + return jsonify({ 'success': True, 'models': models, 'recommended': models[0] if models else '', 'message': f'Found {len(models)} local models' }) - + # Handle Anthropic - no models list API, return verified models directly if provider == 'anthropic': models = list(verified_models) if verified_models else [ @@ -682,31 +508,31 @@ def get_provider_models(): 'claude-3-5-sonnet-latest', 'claude-3-opus-latest', ] - return _reply({ + return jsonify({ 'success': True, 'models': sorted(models), 'recommended': recommended or models[0], 'message': f'{len(models)} verified models' }) - + # For other providers, fetch from API and filter by verified list. # Custom OpenAI-compatible endpoints (LiteLLM, opencode.ai, vLLM, # LocalAI…) often expose `/v1/models` without authentication, so # we only require an api_key when there's no custom base URL to # consult. Issue #11.5 — OpenCode provider Custom Base URL fetch. if not api_key and not (provider == 'openai' and openai_base_url): - return _reply({'success': False, 'models': [], 'message': 'API key required'}) - + return jsonify({'success': False, 'models': [], 'message': 'API key required'}) + from ai_providers import get_provider ai_provider = get_provider( - provider, - api_key=api_key, - model='', + provider, + api_key=api_key, + model='', base_url=openai_base_url if provider == 'openai' else None ) - + if not ai_provider: - return _reply({'success': False, 'models': [], 'message': f'Unknown provider: {provider}'}) + return jsonify({'success': False, 'models': [], 'message': f'Unknown provider: {provider}'}) # Get all models from provider API api_models = ai_provider.list_models() @@ -725,13 +551,13 @@ def get_provider_models(): # so "gpt-4o-mini" as a fallback would be misleading). if verified_models and not is_openai_compat: models = sorted(verified_models) - return _reply({ + return jsonify({ 'success': True, 'models': models, 'recommended': recommended or models[0], 'message': f'{len(models)} verified models (API unavailable)' }) - return _reply({ + return jsonify({ 'success': False, 'models': [], 'message': 'Could not retrieve models. Check your API key and endpoint URL.' @@ -741,7 +567,7 @@ def get_provider_models(): # Custom OpenAI-compatible endpoint: surface every model the # endpoint reports. No verified-list intersection. models = sorted(api_models) - return _reply({ + return jsonify({ 'success': True, 'models': models, 'recommended': models[0] if models else '', @@ -768,29 +594,20 @@ def get_provider_models(): else: # No verified list for this provider, return all from API models = sorted(api_models) - - return _reply({ + + return jsonify({ 'success': True, 'models': models, 'recommended': recommended if recommended in models else (models[0] if models else ''), 'message': f'{len(models)} verified models available' }) - + except Exception as e: - # Outside the _reply closure — build the payload manually so the - # catch-all still carries catalog_meta when a refresh was - # attempted before the crash. - payload = { + return jsonify({ 'success': False, 'models': [], - 'message': f'Error: {str(e)}', - } - try: - if catalog_meta is not None: - payload['catalog_meta'] = catalog_meta - except UnboundLocalError: - pass - return jsonify(payload) + 'message': f'Error: {str(e)}' + }) @notification_bp.route('/api/notifications/test-ai', methods=['POST'])