drop live AI catalog refresh + GH Action

This commit is contained in:
MacRimi
2026-09-02 12:05:53 +02:00
parent c820685241
commit 1bef36fb20
18 changed files with 101 additions and 1261 deletions
@@ -1,16 +0,0 @@
# ProxMenux AI-model verifier — API keys
# Copy this file to .env (gitignored) and fill only the ones you have.
OPENAI_API_KEY=
# Optional: for LiteLLM/MLX/LM Studio/vLLM/LocalAI/Ollama-proxy testing,
# point OPENAI_API_KEY to any non-empty placeholder and set OPENAI_BASE_URL
# to the endpoint's root (without /v1 — the tool appends it).
# OPENAI_BASE_URL=http://localhost:4000
GROQ_API_KEY=
GEMINI_API_KEY=
ANTHROPIC_API_KEY=
OPENROUTER_API_KEY=
@@ -1,37 +0,0 @@
# AI models verifier (public copy)
Standalone verifier used by the daily GitHub Action to refresh
`AppImage/config/verified_ai_models.json`.
The code lives here so the Action can execute it. API keys are read from
GitHub Secrets at run time and never written to disk.
Local dev runs (interactive verifier over your own keys) can keep using
the private copy — `verify.py` is identical.
## What the Action does
Each run:
1. Loads keys from Secrets into environment variables.
2. Runs `verify.py --json-out /tmp/report.json` against every provider that
has a key set.
3. Rewrites `AppImage/config/verified_ai_models.json` with the passing
models, sorted with the recommended one first per provider.
4. Bumps the `_updated` field to the current date.
5. If the file changed, commits directly to `main` as a bot commit.
## Adding provider keys
- Repository → Settings → Secrets and variables → Actions.
- Add each key with the exact name expected by `verify.py`:
`OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GROQ_API_KEY`, `GEMINI_API_KEY`,
`OPENROUTER_API_KEY`.
- Any provider without a key is silently skipped — the Action logs a
warning and continues with the rest.
## Running the Action on demand
The workflow accepts `workflow_dispatch`, so you can trigger a refresh
manually from the Actions tab. Useful when a new model has just been
released upstream and you don't want to wait for the daily cron.
-143
View File
@@ -1,143 +0,0 @@
#!/usr/bin/env python3
"""Apply a verifier report to ``AppImage/config/verified_ai_models.json``.
Reads the machine-readable report emitted by ``verify.py --json-out`` and
merges the passing models into the on-disk catalog. Preserves the
maintainer's editorial curation across three axes:
* ``_exclude``: per-provider list of model IDs (exact match) that must
never appear in the surfaced ``models`` list even when the verifier
passes them. Meant for models that respond correctly to the technical
test but are the wrong fit for notification translation — Arabic-only
bases, Chinese-first fine-tunes, safety-classifier variants,
agentic-only endpoints, legacy dated snapshots, etc.
* ``recommended``: if the current recommendation is still in the
passing (and non-excluded) set, it is preserved. Only when the
previous recommendation disappears (deprecated upstream, or newly
excluded) is a fallback chosen — the fastest passing model.
* ``_note`` / ``_deprecated``: never touched. Those are maintainer
annotations that outlive any single verifier run.
Fail-safe rules:
* Providers absent from the report (no API key configured in the
Action for that run) are left untouched.
* Providers whose report carries an error are left untouched.
* If the ``_exclude`` filter drops every passing model, the block is
left untouched — an empty models list would silently kill the
provider in the UI; keeping the previous list is more forgiving
than shipping "nothing works".
* ``_updated`` bumps to today's date only when the merge actually
changed something. A no-op run leaves the file byte-identical.
Exits 0 when the file is unchanged, 10 when it was updated. The
workflow uses that exit code to decide whether to commit.
"""
from __future__ import annotations
import argparse
import datetime as dt
import fnmatch
import json
import sys
from pathlib import Path
def _load_json(path: Path) -> dict:
with open(path, "r", encoding="utf-8") as fh:
return json.load(fh)
def _save_json(path: Path, data: dict) -> None:
tmp = path.with_suffix(path.suffix + ".tmp")
with open(tmp, "w", encoding="utf-8") as fh:
json.dump(data, fh, indent=2, ensure_ascii=False)
fh.write("\n")
tmp.replace(path)
def _is_excluded(model: str, patterns: list[str]) -> bool:
"""Match a model against the ``_exclude`` list. Supports exact
matches and shell-style globs (``gpt-4o-*``, ``*-2024-*``, ...) so
a provider that periodically publishes dated snapshots can be
covered by a single pattern instead of one entry per date."""
for pat in patterns:
if pat == model or fnmatch.fnmatchcase(model, pat):
return True
return False
def _passing_models(provider_report: dict, exclude: list[str]) -> list[str]:
"""Passing models minus the editorial exclusion list, fastest first."""
passing = [
r for r in provider_report.get("results", [])
if r.get("verdict") == "pass" and not _is_excluded(r.get("model", ""), exclude)
]
passing.sort(key=lambda r: r.get("latency_s", 999))
return [r["model"] for r in passing]
def apply_report(report_path: Path, catalog_path: Path, today: str) -> bool:
"""Rewrite the catalog from the report. Returns True if it changed."""
report = _load_json(report_path)
catalog = _load_json(catalog_path) if catalog_path.exists() else {}
changed = False
for provider_report in report:
name = provider_report.get("provider")
if not name:
continue
if provider_report.get("error"):
print(f"[{name}] skipped — verifier reported error: {provider_report['error']}",
file=sys.stderr)
continue
block = catalog.setdefault(name, {})
exclude = list(block.get("_exclude", []))
passing = _passing_models(provider_report, exclude)
if not passing:
# Either the verifier returned no passes for this provider,
# or every pass got filtered by _exclude. Both cases mean
# "no signal we can trust to overwrite the curated list";
# leaving the block alone is safer than blanking it.
print(f"[{name}] skipped — no passing models after exclude filter",
file=sys.stderr)
continue
prev_models = list(block.get("models", []))
prev_recommended = block.get("recommended", "")
# Preserve the maintainer's choice of recommended when it is
# still valid. Only fall back to fastest when the previous
# value disappeared from the passing set.
recommended = prev_recommended if prev_recommended in passing else passing[0]
if sorted(prev_models) != sorted(passing) or prev_recommended != recommended:
block["models"] = passing
block["recommended"] = recommended
changed = True
print(f"[{name}] updated — {len(passing)} models, recommended={recommended}")
else:
print(f"[{name}] unchanged — {len(passing)} models")
if changed:
catalog["_updated"] = today
_save_json(catalog_path, catalog)
return changed
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--report", required=True, help="verify.py --json-out path")
ap.add_argument("--catalog", required=True,
help="AppImage/config/verified_ai_models.json path")
ap.add_argument("--today", default=None,
help="Override the date written into _updated (YYYY-MM-DD).")
args = ap.parse_args()
today = args.today or dt.datetime.utcnow().strftime("%Y-%m-%d")
changed = apply_report(Path(args.report), Path(args.catalog), today)
return 10 if changed else 0
if __name__ == "__main__":
sys.exit(main())
@@ -1,50 +0,0 @@
"""Standardized test prompt for ProxMenux AI-model verification.
Mirrors the real AI-enrichment use case: take a raw Proxmox system
notification (English, with technical identifiers), translate it into
Spanish, explain in plain terms, and suggest one concrete action. It is
intentionally simple — if a model can't do this, it won't do the real
thing either. Models that pass this test are fine for inclusion in
verified_ai_models.json.
"""
SYSTEM_PROMPT = (
"You are a Proxmox system-notification assistant. "
"When given a raw notification from a Proxmox host, you: "
"(1) translate it into Spanish, "
"(2) explain in 2-3 sentences what the user is seeing and the likely cause, "
"(3) suggest ONE concrete next action. "
"Keep technical identifiers (device paths like /dev/sdd, SMART keywords, "
"ata port numbers, BDFs) in their original form. "
"Respond only in Spanish. Stay under 200 tokens total."
)
# Realistic ProxMenux notification payload: multi-line body with
# SMART/ATA vocabulary and a frequency hint — the exact shape the real
# pipeline emits.
USER_MESSAGE = (
"Event: disk_io_error\n"
"Severity: CRITICAL\n"
"Host: pve-constructor\n"
"Device: /dev/sdd\n"
"SMART status: PASSED\n"
"Summary: 3 I/O event(s) in 5 minutes, disk passed SMART short test\n"
"Sample kernel line: ata4.00: exception Emask 0x0 SAct 0x804000 SErr 0x0 action 0x6\n"
"Frequency: 3 occurrences in 24h, first seen 6h ago"
)
# Common Spanish stopwords. A response missing ALL of these is almost
# certainly not Spanish (or empty/truncated). Cheap heuristic, good
# enough for a coarse pass/fail.
REQUIRED_SPANISH_HINTS = [
" el ", " la ", " los ", " las ", " un ", " una ",
" de ", " del ", " en ", " con ", " que ", " para ",
" es ", " se ", " ha ", " por ", " y ",
]
# Domain keywords — at least one must appear to confirm the model
# actually engaged with the notification instead of replying generically.
DOMAIN_HINTS = [
"disco", "sdd", "smart", "ata", "i/o", "e/s", "error",
"kernel", "proxmox",
]
@@ -1,263 +0,0 @@
"""Self-contained API wrappers for AI-model verification.
Kept independent from the ProxMenux AppImage's ai_providers module so
this tool can live in a private repo with no import coupling to the
public project. Uses only the Python standard library.
"""
from __future__ import annotations
import json
import urllib.error
import urllib.request
from typing import List, Optional
class ProviderError(Exception):
pass
class Provider:
"""Base class. Subclasses implement list_models() and generate()."""
name = "base"
def __init__(self, api_key: str, base_url: Optional[str] = None):
self.api_key = api_key
self.base_url = base_url
def list_models(self) -> List[str]:
raise NotImplementedError
def generate(self, model: str, system: str, user: str,
max_tokens: int = 250, timeout: int = 30) -> str:
raise NotImplementedError
# ── HTTP helpers ────────────────────────────────────────────
# Cloudflare in front of api.groq.com (and probably other providers
# over time) returns 403 "error code: 1010" for the default
# `Python-urllib/3.x` User-Agent — the "browser signature ban" rule.
# A plain identifier is enough to get through; we're not spoofing a
# browser, just avoiding a naive UA fingerprint match.
_USER_AGENT = "ProxMenux-AI-Verifier/1.0"
def _post_json(self, url: str, payload: dict, headers: dict,
timeout: int = 30) -> dict:
data = json.dumps(payload).encode("utf-8")
headers = {**headers, "User-Agent": self._USER_AGENT}
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
with urllib.request.urlopen(req, timeout=timeout) as resp:
return json.loads(resp.read().decode("utf-8"))
def _get_json(self, url: str, headers: dict, timeout: int = 30) -> dict:
headers = {**headers, "User-Agent": self._USER_AGENT}
req = urllib.request.Request(url, headers=headers, method="GET")
with urllib.request.urlopen(req, timeout=timeout) as resp:
return json.loads(resp.read().decode("utf-8"))
# ══════════════════════════════════════════════════════════════════
# OpenAI-compatible family (OpenAI, Groq, OpenRouter, LiteLLM, LM
# Studio, vLLM, LocalAI, etc.). They all speak the same endpoints.
# ══════════════════════════════════════════════════════════════════
class OpenAICompatProvider(Provider):
name = "openai-compat"
default_base = "https://api.openai.com"
models_path = "/v1/models"
chat_path = "/v1/chat/completions"
def _base(self) -> str:
return (self.base_url or self.default_base).rstrip("/")
@staticmethod
def _is_reasoning_model(model: str) -> bool:
"""True for OpenAI reasoning models (o-series + non-chat gpt-5+).
Must be kept in sync with the matching helper in ProxMenux's
openai_provider.py — same rule, same consequence:
- send max_completion_tokens instead of max_tokens
- omit temperature (default is the only accepted value).
"""
m = model.lower()
if len(m) >= 2 and m[0] == "o" and m[1].isdigit():
return True
if m.startswith("gpt-5") and "-chat" not in m:
return True
return False
def list_models(self) -> List[str]:
url = f"{self._base()}{self.models_path}"
headers = {"Authorization": f"Bearer {self.api_key}"}
data = self._get_json(url, headers)
return [m.get("id", "") for m in data.get("data", []) if m.get("id")]
def generate(self, model: str, system: str, user: str,
max_tokens: int = 250, timeout: int = 30) -> str:
url = f"{self._base()}{self.chat_path}"
headers = {
"Authorization": f"Bearer {self.api_key}",
"Content-Type": "application/json",
}
payload = {
"model": model,
"messages": [
{"role": "system", "content": system},
{"role": "user", "content": user},
],
}
if self._is_reasoning_model(model):
# Reasoning models spend budget on internal reasoning by
# default, which yields empty replies at small max_tokens.
# reasoning_effort=minimal keeps that overhead low so the
# whole budget reaches the user, aligned with the short
# translate+explain task ProxMenux uses. Mirror this in
# ProxMenux's openai_provider.py.
payload["max_completion_tokens"] = max_tokens
payload["reasoning_effort"] = "minimal"
else:
payload["max_tokens"] = max_tokens
payload["temperature"] = 0.3
data = self._post_json(url, payload, headers, timeout)
try:
return data["choices"][0]["message"]["content"].strip()
except (KeyError, IndexError) as exc:
raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}")
class OpenAIProvider(OpenAICompatProvider):
name = "openai"
default_base = "https://api.openai.com"
class GroqProvider(OpenAICompatProvider):
name = "groq"
default_base = "https://api.groq.com/openai"
class OpenRouterProvider(OpenAICompatProvider):
name = "openrouter"
default_base = "https://openrouter.ai/api"
# ══════════════════════════════════════════════════════════════════
# Gemini — different endpoint shape, keyed via ?key=... query param.
# ══════════════════════════════════════════════════════════════════
class GeminiProvider(Provider):
name = "gemini"
default_base = "https://generativelanguage.googleapis.com/v1beta"
@staticmethod
def _has_thinking_mode(model: str) -> bool:
"""True for Gemini variants that enable "thinking" by default.
Kept in sync with ProxMenux's gemini_provider.py. 2.5+ pro/flash
and 3.x pro/flash consume output tokens on reasoning, which
yields empty replies when max_tokens is small. We pass
thinkingBudget=0 to disable thinking so the short translate+
explain test sees actual text. Lite variants don't have thinking
enabled and are not flagged here.
"""
m = model.lower()
if "lite" in m:
return False
return m.startswith("gemini-2.5") or m.startswith("gemini-3")
def list_models(self) -> List[str]:
url = f"{self.default_base}/models?key={self.api_key}"
data = self._get_json(url, {})
names = []
for m in data.get("models", []):
raw = m.get("name", "")
if raw.startswith("models/"):
raw = raw[len("models/"):]
# Only keep text-generation capable models.
if "generateContent" in m.get("supportedGenerationMethods", []):
names.append(raw)
return names
def generate(self, model: str, system: str, user: str,
max_tokens: int = 250, timeout: int = 30) -> str:
url = f"{self.default_base}/models/{model}:generateContent?key={self.api_key}"
gen_config = {
"maxOutputTokens": max_tokens,
"temperature": 0.3,
}
if self._has_thinking_mode(model):
gen_config["thinkingConfig"] = {"thinkingBudget": 0}
payload = {
"system_instruction": {"parts": [{"text": system}]},
"contents": [{"parts": [{"text": user}]}],
"generationConfig": gen_config,
}
data = self._post_json(url, payload, {"Content-Type": "application/json"}, timeout)
try:
return data["candidates"][0]["content"]["parts"][0]["text"].strip()
except (KeyError, IndexError) as exc:
raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}")
# ══════════════════════════════════════════════════════════════════
# Anthropic — own schema and headers.
# ══════════════════════════════════════════════════════════════════
class AnthropicProvider(Provider):
name = "anthropic"
default_base = "https://api.anthropic.com"
version = "2023-06-01"
def list_models(self) -> List[str]:
url = f"{self.default_base}/v1/models"
headers = {
"x-api-key": self.api_key,
"anthropic-version": self.version,
}
try:
data = self._get_json(url, headers)
return [m.get("id", "") for m in data.get("data", []) if m.get("id")]
except urllib.error.HTTPError:
# Older keys don't have the endpoint; caller decides what to do.
return []
def generate(self, model: str, system: str, user: str,
max_tokens: int = 250, timeout: int = 30) -> str:
url = f"{self.default_base}/v1/messages"
headers = {
"x-api-key": self.api_key,
"anthropic-version": self.version,
"Content-Type": "application/json",
}
# ProxMenux's real anthropic_provider.py does NOT send `temperature`,
# and Anthropic's newest generation (Claude Sonnet 5, Opus 4.7 /
# 4.8, Fable 5) now rejects it with:
# invalid_request_error: `temperature` is deprecated for this model.
# Omitting it here aligns the verifier with production behaviour so
# the newest-gen models can be reached during verification.
payload = {
"model": model,
"max_tokens": max_tokens,
"system": system,
"messages": [{"role": "user", "content": user}],
}
data = self._post_json(url, payload, headers, timeout)
try:
return data["content"][0]["text"].strip()
except (KeyError, IndexError) as exc:
raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}")
PROVIDERS = {
"openai": OpenAIProvider,
"groq": GroqProvider,
"gemini": GeminiProvider,
"anthropic": AnthropicProvider,
"openrouter": OpenRouterProvider,
}
def make_provider(name: str, api_key: str,
base_url: Optional[str] = None) -> Provider:
cls = PROVIDERS.get(name)
if not cls:
raise ValueError(f"unknown provider: {name}")
return cls(api_key, base_url=base_url)
@@ -1,232 +0,0 @@
#!/usr/bin/env python3
"""ProxMenux AI-model verifier.
Runs a standardized translate+explain test against every model each
provider currently advertises, and emits a per-model verdict so the
verified_ai_models.json list can be refreshed with confidence.
Not packaged with the AppImage — keep this in a private repo alongside
the API keys.
Usage:
cp .env.example .env # fill in the API keys you have
python3 verify.py # test all providers with keys
python3 verify.py --provider groq # just one
python3 verify.py --provider openai --limit 5 # only first 5 models
python3 verify.py --json-out report.json # machine-readable output too
"""
from __future__ import annotations
import argparse
import json
import os
import sys
import time
from pathlib import Path
from typing import Dict, List, Optional, Tuple
from prompts import DOMAIN_HINTS, REQUIRED_SPANISH_HINTS, SYSTEM_PROMPT, USER_MESSAGE
from providers import PROVIDERS, make_provider
# Non-chat model name patterns. Skipping these saves test time and keeps
# the report focused on models that could actually serve notifications.
SKIP_PATTERNS = (
"embedding", "whisper", "tts", "dall-e", "dalle", "image",
"realtime", "audio", "moderation", "search",
"code-search", "text-similarity", "babbage", "davinci",
"curie", "ada", "transcribe",
)
def load_env(env_path: Path) -> Dict[str, str]:
"""Minimal .env loader (avoids a python-dotenv dependency)."""
if not env_path.exists():
return {}
env: Dict[str, str] = {}
for line in env_path.read_text().splitlines():
line = line.strip()
if not line or line.startswith("#") or "=" not in line:
continue
k, v = line.split("=", 1)
env[k.strip()] = v.strip().strip('"').strip("'")
return env
def should_skip_model(model: str) -> bool:
m = model.lower()
return any(p in m for p in SKIP_PATTERNS)
def assess_response(text: str) -> Tuple[str, List[str]]:
"""Classify the model output. Returns (verdict, reasons).
verdict is one of:
- 'pass': Spanish, on-topic, reasonable length
- 'warn': responded but one heuristic failed (borderline)
- 'fail': empty, wrong language, or off-topic
"""
reasons: List[str] = []
if not text or len(text) < 30:
return "fail", ["empty or too short response"]
text_low = " " + text.lower() + " "
spanish_hits = sum(1 for h in REQUIRED_SPANISH_HINTS if h in text_low)
domain_hits = sum(1 for h in DOMAIN_HINTS if h.lower() in text_low)
if spanish_hits < 3:
reasons.append(f"not Spanish ({spanish_hits}/{len(REQUIRED_SPANISH_HINTS)} hints)")
if domain_hits < 1:
reasons.append("did not engage with the domain")
if len(text) > 1500:
reasons.append("response unusually long")
if not reasons:
return "pass", []
# Responded and engaged, but one signal missed → warn (keep in list
# with a caveat; don't auto-include).
if domain_hits >= 1 and spanish_hits >= 1:
return "warn", reasons
return "fail", reasons
def run_model(provider, model: str, timeout: int) -> dict:
t0 = time.time()
try:
out = provider.generate(
model, SYSTEM_PROMPT, USER_MESSAGE,
max_tokens=250, timeout=timeout,
)
latency = time.time() - t0
verdict, reasons = assess_response(out)
return {
"model": model,
"verdict": verdict,
"latency_s": round(latency, 2),
"reasons": reasons,
"sample": (out[:140] if out else "").replace("\n", " "),
"error": None,
}
except Exception as exc: # HTTPError, ProviderError, timeouts
return {
"model": model,
"verdict": "fail",
"latency_s": round(time.time() - t0, 2),
"reasons": [],
"sample": "",
"error": str(exc)[:200],
}
def tag(verdict: str) -> str:
return {"pass": "", "warn": "", "fail": ""}.get(verdict, "?")
def run_provider(name: str, api_key: str, base_url: Optional[str],
timeout: int, limit: Optional[int]) -> dict:
print(f"\n=== {name} ===")
try:
provider = make_provider(name, api_key, base_url=base_url)
models = provider.list_models()
except Exception as exc:
print(f" list_models() failed: {exc}")
return {"provider": name, "error": str(exc), "results": []}
models = [m for m in models if not should_skip_model(m)]
if limit:
models = models[:limit]
if not models:
print(" (no eligible models)")
return {"provider": name, "error": None, "results": []}
print(f" discovered {len(models)} model(s)")
results: List[dict] = []
for m in models:
r = run_model(provider, m, timeout)
results.append(r)
latency = f"{r['latency_s']}s"
if r["error"]:
suffix = f"{r['error']}"
elif r["reasons"]:
suffix = f"{'; '.join(r['reasons'])}"
else:
suffix = ""
print(f" {tag(r['verdict'])} {m:<50} {latency:>6}{suffix}")
return {"provider": name, "error": None, "results": results}
def summarize(all_results: List[dict]) -> None:
print("\n" + "=" * 64)
print("Suggested verified_ai_models.json entries (passing models only)")
print("=" * 64)
any_output = False
for pr in all_results:
if pr.get("error"):
continue
passed = [r for r in pr["results"] if r["verdict"] == "pass"]
if not passed:
continue
any_output = True
passed_sorted = sorted(passed, key=lambda x: x["latency_s"])
print(f'\n "{pr["provider"]}": {{')
print(' "models": [')
for r in passed_sorted:
print(f' "{r["model"]}",')
print(" ],")
print(f' "recommended": "{passed_sorted[0]["model"]}"')
print(" },")
warn_total = sum(
1 for pr in all_results for r in pr["results"]
if r["verdict"] == "warn"
)
if warn_total:
print(f"\n Note: {warn_total} model(s) came back as ⚠ (warn) — review those manually.")
if not any_output:
print("\n (no models passed; check keys and network)")
def main() -> int:
ap = argparse.ArgumentParser(
description="Verify AI models for ProxMenux notification enrichment."
)
ap.add_argument("--env", default=".env", help=".env file path")
ap.add_argument("--provider", action="append", default=[],
help="run a specific provider (repeat to add more)")
ap.add_argument("--timeout", type=int, default=30,
help="seconds per request (default: 30)")
ap.add_argument("--limit", type=int, default=None,
help="test at most N models per provider (debug)")
ap.add_argument("--json-out", default=None,
help="write machine-readable report to this path")
args = ap.parse_args()
env = {**os.environ, **load_env(Path(args.env))}
provider_list = args.provider or list(PROVIDERS.keys())
tested: List[dict] = []
for name in provider_list:
if name not in PROVIDERS:
print(f"unknown provider: {name}", file=sys.stderr)
continue
key_var = f"{name.upper()}_API_KEY"
url_var = f"{name.upper()}_BASE_URL"
api_key = env.get(key_var, "")
base_url = env.get(url_var) or None
if not api_key:
print(f"\n=== {name} ===\n skipped — {key_var} not set")
continue
tested.append(run_provider(name, api_key, base_url, args.timeout, args.limit))
summarize(tested)
if args.json_out:
Path(args.json_out).write_text(json.dumps(tested, indent=2))
print(f"\nReport written to {args.json_out}")
return 0
if __name__ == "__main__":
sys.exit(main())
-103
View File
@@ -1,103 +0,0 @@
name: Verify AI models catalog
# Runs the AI-model verifier and commits any changes to
# AppImage/config/verified_ai_models.json on the same branch the run
# was launched from.
#
# GitHub only fires `on: schedule` from the default branch, so the
# daily cron always runs against main and keeps stable users fresh.
# When a beta cycle needs its own refresh on develop, use the
# "Run workflow" button on the Actions tab and pick develop from the
# branch selector — the same YAML then checks out develop, runs the
# verifier and commits back to develop. Cross-branch pushes never
# happen: each run only touches the branch it started on.
#
# The verifier code lives at .github/scripts/ai-models-verifier/ and
# reads API keys from repository Secrets. Any provider without a key
# is skipped silently — the workflow keeps going with the rest.
on:
schedule:
- cron: '0 4 * * *' # 04:00 UTC every day — cron always fires from main
workflow_dispatch: # manual trigger — branch is picked in the UI
permissions:
contents: write
concurrency:
# Keyed by branch so a manual develop run does not collide with the
# scheduled main run — each branch gets its own serialisation lane.
group: verify-ai-models-${{ github.ref_name }}
cancel-in-progress: false
jobs:
verify:
runs-on: ubuntu-latest
timeout-minutes: 20
steps:
- name: Check out the branch this run belongs to
uses: actions/checkout@v4
with:
# `github.ref_name` resolves to main for the cron and to the
# branch selected in the dispatch UI otherwise. The same
# value is used again below when we push, so every run is
# symmetric: checkout X → refresh → push X.
ref: ${{ github.ref_name }}
token: ${{ secrets.GITHUB_TOKEN }}
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.12'
- name: Run verifier
working-directory: .github/scripts/ai-models-verifier
env:
# API keys — each is optional. verify.py silently skips any
# provider whose *_API_KEY env var is empty, so the workflow
# runs even when only a subset of keys is configured.
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
GROQ_API_KEY: ${{ secrets.GROQ_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
# Optional base URLs for custom-endpoint providers.
OPENAI_BASE_URL: ${{ secrets.OPENAI_BASE_URL }}
run: |
python3 verify.py --json-out /tmp/report.json || true
if [ ! -s /tmp/report.json ]; then
echo "Verifier produced no report — bailing"
exit 1
fi
- name: Apply report to catalog
id: apply
working-directory: .
run: |
set +e
python3 .github/scripts/ai-models-verifier/apply.py \
--report /tmp/report.json \
--catalog AppImage/config/verified_ai_models.json
code=$?
set -e
case "$code" in
0) echo "changed=false" >> "$GITHUB_OUTPUT" ;;
10) echo "changed=true" >> "$GITHUB_OUTPUT" ;;
*) echo "apply.py exited with $code"; exit "$code" ;;
esac
- name: Commit and push back to the same branch
if: steps.apply.outputs.changed == 'true'
run: |
git config user.name "proxmenux-bot"
git config user.email "proxmenux-bot@users.noreply.github.com"
git add AppImage/config/verified_ai_models.json
git commit -m "chore(ai-models): daily catalog refresh"
# Push to the branch this run started on — same ref used at
# checkout above, so the operation is symmetric regardless of
# whether cron (main) or dispatch (any branch) triggered it.
git push origin HEAD:${{ github.ref_name }}
- name: Report unchanged
if: steps.apply.outputs.changed != 'true'
run: echo "Catalog already up to date — nothing to commit."
+2 -65
View File
@@ -341,16 +341,6 @@ export function NotificationSettings() {
const [aiTestResult, setAiTestResult] = useState<{ success: boolean; message: string; model?: string } | null>(null)
const [providerModels, setProviderModels] = useState<string[]>([])
const [loadingProviderModels, setLoadingProviderModels] = useState(false)
// Metadata returned by the backend after the catalog refresh — used
// to render the "Last update" line and toast the diff.
const [catalogMeta, setCatalogMeta] = useState<{
success: boolean
message: string
changed: boolean
previous_updated: string | null
new_updated: string | null
diff: Record<string, { added: string[]; removed: string[] }>
} | null>(null)
const [showCustomPromptInfo, setShowCustomPromptInfo] = useState(false)
const [editingCustomPrompt, setEditingCustomPrompt] = useState(false)
const [customPromptDraft, setCustomPromptDraft] = useState("")
@@ -1003,20 +993,7 @@ export function NotificationSettings() {
setLoadingProviderModels(true)
try {
const data = await fetchApi<{
success: boolean
models: string[]
recommended: string
message: string
catalog_meta?: {
success: boolean
message: string
changed: boolean
previous_updated: string | null
new_updated: string | null
diff: Record<string, { added: string[]; removed: string[] }>
} | null
}>("/api/notifications/provider-models", {
const data = await fetchApi<{ success: boolean; models: string[]; recommended: string; message: string }>("/api/notifications/provider-models", {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
@@ -1024,17 +1001,8 @@ export function NotificationSettings() {
api_key: apiKey,
ollama_url: config.ai_ollama_url,
openai_base_url: config.ai_openai_base_url,
// Every click pulls the latest verified catalog from GitHub
// before the intersection runs, so a user with an outdated
// AppImage still sees today's model list. The backend keeps
// the local file when the fetch fails, so an offline node
// simply doesn't refresh but still resolves models.
refresh_catalog: true,
}),
})
if (data.catalog_meta) {
setCatalogMeta(data.catalog_meta)
}
if (data.success && data.models && data.models.length > 0) {
setProviderModels(data.models)
// Auto-select recommended model if current selection is empty or not in the list
@@ -2451,12 +2419,7 @@ export function NotificationSettings() {
) : (
<>
<RefreshCw className="h-4 w-4 mr-1" />
{/* Label swaps once we have models loaded — "Load" reads
right on the empty state, "Update" reads right after
the first pull. Same action underneath either way. */}
{t(providerModels.length === 0
? "settings.notifications.ai.load"
: "settings.notifications.ai.updateModels")}
{t("settings.notifications.ai.load")}
</>
)}
</Button>
@@ -2464,32 +2427,6 @@ export function NotificationSettings() {
{providerModels.length > 0 && (
<p className="text-xs text-green-500">{t("settings.notifications.ai.modelsAvailable", { count: providerModels.length })}</p>
)}
{/* Catalog freshness line — surfaces the `_updated` date of the
verified_ai_models.json currently in use, plus a compact diff
summary from the most recent GitHub pull so the user knows
what just happened without opening a modal. */}
{catalogMeta && (
<div className="text-[11px] text-muted-foreground space-y-0.5">
<div>
{t("settings.notifications.ai.catalogLastUpdated", {
date: catalogMeta.new_updated || catalogMeta.previous_updated || "—",
})}
</div>
{catalogMeta.changed && catalogMeta.success && Object.keys(catalogMeta.diff).length > 0 && (
<div className="text-blue-400">
{t("settings.notifications.ai.catalogChanged", {
added: Object.values(catalogMeta.diff).reduce((n, d) => n + d.added.length, 0),
removed: Object.values(catalogMeta.diff).reduce((n, d) => n + d.removed.length, 0),
})}
</div>
)}
{!catalogMeta.success && catalogMeta.message && (
<div className="text-amber-400">
{t("settings.notifications.ai.catalogFetchFailed", { reason: catalogMeta.message })}
</div>
)}
</div>
)}
</div>
{/* Prompt Mode section */}
+50 -88
View File
@@ -1,123 +1,85 @@
{
"_description": "Verified AI models for ProxMenux notifications. Only models listed here will be shown to users. Models are tested to work with the chat/completions API format.",
"_updated": "2026-09-02",
"_verifier": "Refreshed by .github/workflows/verify-ai-models.yml (daily). The workflow runs .github/scripts/ai-models-verifier/verify.py against every provider whose API key is configured in repository Secrets, then applies the report via apply.py — which honours per-provider `_exclude` lists. `_exclude` is intentionally minimal: it only drops models that are technically incapable of generating a chat completion for a normal prompt (safety classifiers, agentic-only endpoints, meta-routers, wrong modalities). Everything else the verifier passes is surfaced — including language-specialised models (Arabic, Chinese, ...), reasoning models, legacy families and dated snapshots — so a user with a specific need can still pick the model that fits. Manually re-run from the Actions tab (any branch) when a new model needs to be picked up out of cycle.",
"_updated": "2026-07-14",
"_verifier": "Refreshed with tools/ai-models-verifier (private). Re-run before each ProxMenux release to keep the list current. The verifier and ProxMenux share the same reasoning/thinking-model handlers so their verdicts stay aligned with runtime behaviour.",
"groq": {
"models": [
"allam-2-7b",
"qwen/qwen3.8-27b",
"openai/gpt-oss-120b"
"llama-3.3-70b-versatile",
"llama-3.1-8b-instant",
"meta-llama/llama-4-scout-17b-16e-instruct",
"openai/gpt-oss-120b",
"openai/gpt-oss-20b"
],
"recommended": "allam-2-7b",
"_exclude": [
"openai/gpt-oss-safeguard-*",
"groq/compound",
"groq/compound-*"
],
"_note": "`_exclude` covers models the verifier may technically pass but that do not produce a usable chat completion: openai/gpt-oss-safeguard-* is a safety classifier (returns a category, not free text); groq/compound* is an agentic system that expects multi-step tool use, not a plain prompt."
"recommended": "llama-3.3-70b-versatile",
"_note": "Verified functionally 2026-07-14 with the Groq API (15 models discovered, 9 passed). Legacy llama-3.1-70b-versatile / llama3-70b-8192 / llama3-8b-8192 / mixtral-8x7b-32768 / gemma2-9b-it removed (retired upstream). llama-4-scout added (current-gen Llama 4, 0.47s). openai/gpt-oss-120b / gpt-oss-20b confirmed. Passing but excluded: allam-2-7b (Arabic-focused), qwen/qwen3-32b (Chinese-first, unreliable Spanish output), openai/gpt-oss-safeguard-20b (safety-classifier variant), groq/compound-mini (agentic system, wrong fit for notification translation)."
},
"gemini": {
"models": [
"gemini-2.5-flash-lite",
"gemini-3.1-flash-lite-preview",
"gemini-3.1-flash-lite",
"gemini-flash-lite-latest",
"gemini-3.5-flash-lite",
"gemini-2.5-flash-lite",
"gemini-2.5-flash",
"gemini-3.5-flash",
"gemini-3.1-flash-lite",
"gemini-3-flash-preview",
"gemma-4-26b-a4b-it"
"gemini-3.5-flash"
],
"recommended": "gemini-2.5-flash-lite",
"_exclude": [
"gemini-embedding-*",
"gemini-*-pro*",
"gemini-*-thinking*"
],
"_deprecated": [
"gemini-2.0-flash",
"gemini-2.0-flash-lite",
"gemini-1.5-flash",
"gemini-1.0-pro",
"gemini-pro"
],
"_note": "`_exclude` drops embeddings (wrong modality) and Pro / thinking variants that reject `thinkingConfig.thinkingBudget: 0` and therefore never return a visible completion within a reasonable token budget — technical failure with our current provider config."
"_note": "Verified 2026-07-13. gemini-flash-lite-latest now passes consistently (1.6s) and is fastest, but gemini-2.5-flash-lite remains recommended because 'latest' aliases can drift over time. gemini-3.1-flash-lite is the stable successor to 3-flash-preview. Pro variants continue to reject thinkingBudget=0 and are overkill for notification translation.",
"_deprecated": ["gemini-2.0-flash", "gemini-2.0-flash-lite", "gemini-1.5-flash", "gemini-1.0-pro", "gemini-pro"]
},
"openai": {
"models": [
"gpt-4.1-nano-2025-04-14",
"gpt-4.1-nano",
"gpt-4.1-2025-04-14",
"gpt-4.1",
"gpt-4o-2024-05-13",
"gpt-4o-2024-11-20",
"gpt-5-nano",
"gpt-5-nano-2025-08-07",
"gpt-3.5-turbo-1106",
"gpt-3.5-turbo-0125",
"gpt-4o-mini-2024-07-18",
"gpt-4.1-mini-2025-04-14",
"gpt-4o-mini",
"gpt-3.5-turbo-16k",
"gpt-4.1-mini",
"gpt-3.5-turbo",
"gpt-4o-mini",
"gpt-4.1",
"gpt-4o",
"gpt-4",
"gpt-4-0613",
"gpt-4-turbo",
"gpt-4-turbo-2024-04-09",
"gpt-4o-2024-08-06"
"gpt-5-chat-latest",
"gpt-5-nano"
],
"recommended": "gpt-4.1-nano",
"_exclude": [
"gpt-4o-audio*",
"gpt-4o-realtime*",
"gpt-4o-search*",
"gpt-4o-transcribe*",
"gpt-4o-mini-audio*",
"gpt-4o-mini-realtime*",
"gpt-4o-mini-search*",
"gpt-4o-mini-transcribe*",
"gpt-4o-mini-tts",
"computer-use-*"
],
"_note": "`_exclude` covers wrong-modality variants (audio, realtime, search, transcribe, tts) that cannot handle a plain notification-translation prompt, plus computer-use which requires an agent loop. All other OpenAI chat/completion models are surfaced — including legacy families (gpt-3.5, gpt-4), reasoning models (o-series, gpt-5.x non-chat) and dated snapshots — so users can pick by their own criteria (cost, quality, reproducibility). openai_provider.py already handles reasoning models via max_completion_tokens + reasoning_effort=minimal."
"_note": "Verified 2026-07-13. gpt-5.4-nano / gpt-5.4-mini removed (HTTP 400 — provider params rejected). gpt-5-nano added (2.0s, current-gen fast). Reasoning models (o-series, gpt-5/5.1/5.2 non-chat variants) are supported by openai_provider.py via max_completion_tokens + reasoning_effort=minimal, but not listed here: their latency is higher and they do not improve translation quality for notifications. Add specific reasoning IDs to this list only if a user explicitly wants them."
},
"anthropic": {
"models": [
"claude-haiku-4-5-20251001",
"claude-haiku-4-5",
"claude-sonnet-5",
"claude-opus-4-8",
"claude-opus-4-5-20251101",
"claude-opus-4-7",
"claude-fable-5",
"claude-sonnet-4-5-20250929",
"claude-fable-5-1",
"claude-sonnet-4-6",
"claude-opus-4-6",
"claude-sonnet-4-6"
"claude-fable-5"
],
"recommended": "claude-haiku-4-5-20251001",
"_exclude": [],
"_note": "No technical exclusions — every Claude generation returns free-text completions for a plain prompt. The verifier tests each model listed under `models`; if a specific ID stops working upstream it simply drops out of the passing set. Anthropic does not expose a public models-list API, so new models must be added to `models` manually before the verifier can test them."
"recommended": "claude-haiku-4-5",
"_note": "Verified 2026-07-13 with all 10 discovered models passing after aligning the verifier with anthropic_provider.py (temperature omitted — newest generations reject it with 'temperature is deprecated for this model'). Legacy claude-3-5-haiku-latest / claude-3-5-sonnet-latest / claude-3-opus-latest removed (deprecated upstream, not in the Models API). haiku-4-5 is the sweet spot for notification translation (3.6s, $1/$5 per MTok); sonnet-5 for slightly richer output (3.1s, $3/$15); opus-4-8 / fable-5 for demanding cases."
},
"openrouter": {
"models": [
"minimax/minimax-m3:free",
"poolside/laguna-s-2.1:free"
"meta-llama/llama-3.3-70b-instruct",
"meta-llama/llama-3.1-70b-instruct",
"meta-llama/llama-3.1-8b-instruct",
"meta-llama/llama-4-scout",
"anthropic/claude-haiku-4.5",
"anthropic/claude-sonnet-4.6",
"google/gemini-2.5-flash-lite",
"google/gemini-2.5-flash",
"openai/gpt-4o-mini",
"mistralai/mistral-small-3.2-24b-instruct",
"nvidia/nemotron-3-super-120b-a12b:free",
"google/gemma-4-26b-a4b-it:free",
"nvidia/nemotron-nano-12b-v2-vl:free",
"nvidia/nemotron-3-nano-30b-a3b:free",
"poolside/laguna-s-2.1:free",
"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free",
"openai/gpt-oss-20b:free"
],
"recommended": "minimax/minimax-m3:free",
"_exclude": [
"openrouter/*",
"*/*-audio*",
"*/*-audio-*",
"*/*-tts*",
"*/*-whisper*",
"*/*-embed*",
"*/*-embedding*",
"*/*-image*",
"*/*-vision-only*"
],
"_note": "OpenRouter aggregates hundreds of models; the free-tier variants (:free suffix) are intentionally supported per user request and never blocked. `_exclude` covers only meta-routers (`openrouter/free`, `openrouter/auto` — they route dynamically to something else, so their behaviour is not the model the user picked) and wrong-modality models (audio, tts, whisper, embeddings, image, vision-only). Everything else — chat models across every family, language and price tier — is surfaced so the user can pick the fit."
"recommended": "meta-llama/llama-3.3-70b-instruct",
"_note": "Paid tier verified functionally 2026-07-14 with the OpenRouter API — all 10 curated candidates pass the Spanish-translation notification test. Fastest: llama-4-scout (0.51s), gemini-2.5-flash-lite (1.14s), gemini-2.5-flash (1.94s), llama-3.3-70b-instruct (2.29s), claude-haiku-4.5 (2.71s). Free tier verified 2026-08-17 — 7 :free models pass and are appended, ordered by latency: nemotron-3-super-120b-a12b (3.5s), gemma-4-26b-a4b-it (4.1s), nemotron-nano-12b-v2-vl (5.3s), nemotron-3-nano-30b-a3b (5.8s), laguna-s-2.1 (8.3s), nemotron-3-nano-omni-30b-a3b-reasoning (10.7s), gpt-oss-20b (12.2s). Free-tier rate limits (~20 req/min shared across all OpenRouter free users on that model) may cause 429 in high-traffic windows — usable for occasional notification translation, not for high-volume automation. recommended kept as llama-3.3-70b for capability/latency balance; llama-4-scout is a faster alternative worth considering as recommended after a broader release."
},
"ollama": {
"_note": "Ollama models are local, we don't filter them. User manages their own models.",
"models": [],
+1 -5
View File
@@ -1922,11 +1922,7 @@
"gemini": "Es ist eine kostenlose Stufe mit einem guten Preis-Leistungs-Verhältnis verfügbar.",
"ollama": "Verwendet Modelle auf Ihrem Ollama-Server. Völlig lokal, privat und kostenlos nutzbar.",
"openrouter": "Zugriff auf mehr als 100 Modelle über einen API-Schlüssel."
},
"updateModels": "Aktualisieren",
"catalogLastUpdated": "Katalog zuletzt aktualisiert: {date}",
"catalogChanged": "Katalog aktualisiert — {added} hinzugefügt, {removed} entfernt",
"catalogFetchFailed": "Katalog konnte nicht aktualisiert werden: {reason}. Lokale Kopie wird verwendet."
}
},
"telegramGuide": {
"title": "Anleitung zur Einrichtung des Telegram-Bots",
+1 -5
View File
@@ -1921,11 +1921,7 @@
"gemini": "A free tier is available, with a good quality-to-price ratio.",
"ollama": "Uses models on your Ollama server. Fully local, private and free to run.",
"openrouter": "Access to more than 100 models through one API key."
},
"updateModels": "Update",
"catalogLastUpdated": "Verified catalog last updated: {date}",
"catalogChanged": "Catalog refreshed — {added} added, {removed} removed",
"catalogFetchFailed": "Could not refresh catalog: {reason}. Using local copy."
}
},
"telegramGuide": {
"title": "Telegram bot setup guide",
+1 -5
View File
@@ -1922,11 +1922,7 @@
"gemini": "Hay disponible un nivel gratuito, con una buena relación calidad-precio.",
"ollama": "Utiliza modelos en su servidor Ollama. Totalmente local, privado y gratuito.",
"openrouter": "Acceso a más de 100 modelos a través de una clave API."
},
"updateModels": "Actualizar",
"catalogLastUpdated": "Última actualización del catálogo verificado: {date}",
"catalogChanged": "Catálogo actualizado — {added} añadidos, {removed} retirados",
"catalogFetchFailed": "No se pudo actualizar el catálogo: {reason}. Se usa la copia local."
}
},
"telegramGuide": {
"title": "Guía de configuración del bot de Telegram",
+1 -5
View File
@@ -1922,11 +1922,7 @@
"gemini": "Un niveau gratuit est disponible, avec un bon rapport qualité-prix.",
"ollama": "Utilise des modèles sur votre serveur Ollama. Entièrement local, privé et gratuit.",
"openrouter": "Accès à plus de 100 modèles via une seule clé API."
},
"updateModels": "Actualiser",
"catalogLastUpdated": "Dernière mise à jour du catalogue vérifié : {date}",
"catalogChanged": "Catalogue mis à jour — {added} ajoutés, {removed} retirés",
"catalogFetchFailed": "Impossible d'actualiser le catalogue : {reason}. Copie locale utilisée."
}
},
"telegramGuide": {
"title": "Guide de configuration du robot Telegram",
+1 -5
View File
@@ -1922,11 +1922,7 @@
"gemini": "È disponibile un livello gratuito, con un buon rapporto qualità-prezzo.",
"ollama": "Utilizza i modelli sul tuo server Ollama. Completamente locale, privato e gratuito.",
"openrouter": "Accesso a più di 100 modelli tramite una chiave API."
},
"updateModels": "Aggiorna",
"catalogLastUpdated": "Ultimo aggiornamento del catalogo verificato: {date}",
"catalogChanged": "Catalogo aggiornato — {added} aggiunti, {removed} rimossi",
"catalogFetchFailed": "Impossibile aggiornare il catalogo: {reason}. Verrà usata la copia locale."
}
},
"telegramGuide": {
"title": "Guida alla configurazione del bot di Telegram",
+1 -5
View File
@@ -1922,11 +1922,7 @@
"gemini": "Um nível gratuito está disponível, com uma boa relação qualidade/preço.",
"ollama": "Usa modelos em seu servidor Ollama. Totalmente local, privado e de operação gratuita.",
"openrouter": "Acesso a mais de 100 modelos através de uma chave API."
},
"updateModels": "Atualizar",
"catalogLastUpdated": "Última atualização do catálogo verificado: {date}",
"catalogChanged": "Catálogo atualizado — {added} adicionados, {removed} removidos",
"catalogFetchFailed": "Não foi possível atualizar o catálogo: {reason}. A usar a cópia local."
}
},
"telegramGuide": {
"title": "Guia de configuração do bot do Telegram",
+1 -5
View File
@@ -1921,11 +1921,7 @@
"gemini": "Ponúka bezplatnú úroveň a dobrý pomer kvality a ceny.",
"ollama": "Používa modely na vašom Ollama serveri. Beží lokálne, súkromne a bez poplatkov.",
"openrouter": "Prístup k viac než 100 modelom cez jeden API kľúč."
},
"updateModels": "Aktualizovať",
"catalogLastUpdated": "Posledná aktualizácia overeného katalógu: {date}",
"catalogChanged": "Katalóg aktualizovaný — {added} pridaných, {removed} odstránených",
"catalogFetchFailed": "Katalóg nebolo možné aktualizovať: {reason}. Používa sa lokálna kópia."
}
},
"telegramGuide": {
"title": "Nastavenie Telegram bota",
+1 -5
View File
@@ -1922,11 +1922,7 @@
"gemini": "En gratis nivå är tillgänglig, med ett bra förhållande mellan kvalitet och pris.",
"ollama": "Använder modeller på din Ollama-server. Helt lokalt, privat och gratis att köra.",
"openrouter": "Tillgång till mer än 100 modeller via en API-nyckel."
},
"updateModels": "Uppdatera",
"catalogLastUpdated": "Verifierad katalog senast uppdaterad: {date}",
"catalogChanged": "Katalog uppdaterad — {added} tillagda, {removed} borttagna",
"catalogFetchFailed": "Katalogen kunde inte uppdateras: {reason}. Använder lokal kopia."
}
},
"telegramGuide": {
"title": "Installationsguide för Telegram bot",
+29 -212
View File
@@ -402,190 +402,32 @@ def test_notification():
return jsonify({'error': f'Internal error ({type(e).__name__})'}), 500
_VERIFIED_MODELS_REMOTE_URL = (
"https://raw.githubusercontent.com/MacRimi/ProxMenux/main/"
"AppImage/config/verified_ai_models.json"
)
def _resolve_verified_models_path() -> Path:
"""Locate the on-disk verified_ai_models.json used at runtime.
AppImage layout keeps scripts and config under /usr/bin/; the dev
tree has them one level apart. We probe both and return whichever
exists so `load_verified_models` and the refresh helper agree on the
same file otherwise a refresh writing to one path and a read
hitting the other would silently do nothing.
"""
script_dir = Path(__file__).parent
candidate = script_dir / 'config' / 'verified_ai_models.json'
if candidate.exists():
return candidate
dev_candidate = script_dir.parent / 'config' / 'verified_ai_models.json'
if dev_candidate.exists():
return dev_candidate
# No file exists yet — return the AppImage-shaped path so a fresh
# refresh has somewhere to write. The parent dir is created if needed.
candidate.parent.mkdir(parents=True, exist_ok=True)
return candidate
def load_verified_models():
"""Load verified models from config file.
Returns {} if the file is missing or unreadable; callers already
handle the empty case by returning provider defaults or the API's
unfiltered list.
Checks multiple paths:
1. Same directory as script (AppImage: /usr/bin/config/)
2. Parent directory config folder (dev: AppImage/config/)
"""
try:
path = _resolve_verified_models_path()
if path.exists():
with open(path, 'r') as f:
# Try AppImage path first (scripts and config both in /usr/bin/)
script_dir = Path(__file__).parent
config_path = script_dir / 'config' / 'verified_ai_models.json'
if not config_path.exists():
# Try development path (AppImage/scripts/ -> AppImage/config/)
config_path = script_dir.parent / 'config' / 'verified_ai_models.json'
if config_path.exists():
with open(config_path, 'r') as f:
return json.load(f)
print(f"[flask_notification_routes] Config not found at {path}")
else:
print(f"[flask_notification_routes] Config not found at {config_path}")
except Exception as e:
print(f"[flask_notification_routes] Failed to load verified models: {e}")
return {}
def _catalog_signature(catalog: dict) -> dict:
"""Reduce a catalog to `{provider: [sorted models]}`.
Used to diff the local vs the remote catalog without dragging the
unrelated `_note` / `_deprecated` / `_updated` metadata into the
comparison.
"""
sig = {}
for provider, block in catalog.items():
if isinstance(block, dict) and isinstance(block.get('models'), list):
sig[provider] = sorted(str(m) for m in block['models'])
return sig
def refresh_verified_models_from_github() -> dict:
"""Fetch the canonical catalog from the main branch and overwrite the
local copy in-place.
Returns a dict shaped for the API response:
{
'success': True|False,
'message': '...',
'changed': bool,
'previous_updated': 'YYYY-MM-DD' | None,
'new_updated': 'YYYY-MM-DD' | None,
'diff': { <provider>: {'added': [...], 'removed': [...]} },
}
Never touches the local file when the remote fetch fails the
installed catalog remains authoritative in the offline / degraded
case, which is the deliberate behaviour the UI relies on to show
'the catalog is still active' after a failed refresh.
"""
import urllib.request
import urllib.error
result = {
'success': False,
'message': '',
'changed': False,
'previous_updated': None,
'new_updated': None,
'diff': {},
}
try:
req = urllib.request.Request(_VERIFIED_MODELS_REMOTE_URL, method='GET')
req.add_header('User-Agent', 'ProxMenux-Monitor/1.1')
req.add_header('Accept', 'application/json')
with urllib.request.urlopen(req, timeout=15) as resp:
raw = resp.read().decode('utf-8')
remote = json.loads(raw)
except urllib.error.HTTPError as e:
result['message'] = f'GitHub returned HTTP {e.code}'
return result
except urllib.error.URLError as e:
result['message'] = f'Could not reach GitHub: {e.reason}'
return result
except (json.JSONDecodeError, ValueError) as e:
result['message'] = f'Remote catalog is not valid JSON: {e}'
return result
except Exception as e:
result['message'] = f'Refresh failed: {type(e).__name__}: {e}'
return result
if not isinstance(remote, dict) or not any(
isinstance(remote.get(k), dict) and 'models' in remote[k]
for k in remote.keys()
if not k.startswith('_')
):
result['message'] = 'Remote catalog has an unexpected shape'
return result
local = load_verified_models() or {}
prev_sig = _catalog_signature(local)
new_sig = _catalog_signature(remote)
diff = {}
for provider in sorted(set(prev_sig) | set(new_sig)):
prev_models = set(prev_sig.get(provider, []))
new_models = set(new_sig.get(provider, []))
added = sorted(new_models - prev_models)
removed = sorted(prev_models - new_models)
if added or removed:
diff[provider] = {'added': added, 'removed': removed}
result['previous_updated'] = local.get('_updated')
result['new_updated'] = remote.get('_updated')
result['changed'] = bool(diff) or (result['previous_updated'] != result['new_updated'])
result['diff'] = diff
# Only touch disk when something actually changed to keep mtimes
# meaningful and avoid a spurious modification in backup diffs.
if result['changed']:
try:
path = _resolve_verified_models_path()
tmp = path.with_suffix(path.suffix + '.tmp')
with open(tmp, 'w') as f:
json.dump(remote, f, indent=2, ensure_ascii=False)
f.write('\n')
tmp.replace(path)
result['success'] = True
result['message'] = 'Catalog updated'
except Exception as e:
result['success'] = False
result['message'] = f'Downloaded but failed to save: {type(e).__name__}: {e}'
return result
else:
result['success'] = True
result['message'] = 'Catalog already up to date'
return result
@notification_bp.route('/api/notifications/refresh-model-catalog', methods=['POST'])
@require_auth
def refresh_model_catalog():
"""Pull the verified_ai_models.json from ProxMenux/main on GitHub and
overwrite the local copy. Returns the diff so the UI can toast a
meaningful summary (added / removed models per provider) instead of a
generic 'refreshed' message.
Standalone endpoint so an admin can refresh without also fetching
provider models; the provider-models call carries a shortcut flag
for the common 'refresh + load' click from the Notifications UI.
"""
try:
return jsonify(refresh_verified_models_from_github())
except Exception as e:
return jsonify({
'success': False,
'message': f'Refresh failed: {type(e).__name__}: {e}',
'changed': False,
'previous_updated': None,
'new_updated': None,
'diff': {},
}), 500
@notification_bp.route('/api/notifications/provider-models', methods=['POST'])
@require_auth
def get_provider_models():
@@ -617,25 +459,9 @@ def get_provider_models():
api_key = _resolve_masked_api_key(provider, data.get('api_key', ''))
ollama_url = data.get('ollama_url', 'http://localhost:11434')
openai_base_url = data.get('openai_base_url', '')
# `refresh_catalog=true` in the request body triggers a GitHub
# pull of verified_ai_models.json before the intersection runs.
# The Notifications UI passes this on every Load / Update click
# so the catalog stays fresh without a separate button.
refresh_catalog = bool(data.get('refresh_catalog', False))
catalog_meta = None
if refresh_catalog:
catalog_meta = refresh_verified_models_from_github()
def _reply(payload, status=200):
# Every response from this endpoint carries `catalog_meta`
# when a refresh was requested, so the UI can render the
# diff toast + "last updated" line from a single roundtrip.
if catalog_meta is not None:
payload = {**payload, 'catalog_meta': catalog_meta}
return jsonify(payload), status
if not provider:
return _reply({'success': False, 'models': [], 'message': 'Provider not specified'})
return jsonify({'success': False, 'models': [], 'message': 'Provider not specified'})
# SSRF guard before we touch the URL. Ollama is local-by-design so
# loopback is allowed there; OpenAI base URL must be a real external
@@ -643,11 +469,11 @@ def get_provider_models():
if provider == 'ollama':
ok, err = validate_external_url(ollama_url, allow_loopback=True)
if not ok:
return _reply({'success': False, 'models': [], 'message': f'Invalid ollama_url: {err}'}, 400)
return jsonify({'success': False, 'models': [], 'message': f'Invalid ollama_url: {err}'}), 400
if provider == 'openai' and openai_base_url:
ok, err = validate_external_url(openai_base_url, allow_loopback=False)
if not ok:
return _reply({'success': False, 'models': [], 'message': f'Invalid openai_base_url: {err}'}, 400)
return jsonify({'success': False, 'models': [], 'message': f'Invalid openai_base_url: {err}'}), 400
# Load verified models config
verified_config = load_verified_models()
@@ -668,7 +494,7 @@ def get_provider_models():
result = json.loads(resp.read().decode('utf-8'))
models = [m.get('name', '') for m in result.get('models', []) if m.get('name')]
models = sorted(models)
return _reply({
return jsonify({
'success': True,
'models': models,
'recommended': models[0] if models else '',
@@ -682,7 +508,7 @@ def get_provider_models():
'claude-3-5-sonnet-latest',
'claude-3-opus-latest',
]
return _reply({
return jsonify({
'success': True,
'models': sorted(models),
'recommended': recommended or models[0],
@@ -695,7 +521,7 @@ def get_provider_models():
# we only require an api_key when there's no custom base URL to
# consult. Issue #11.5 — OpenCode provider Custom Base URL fetch.
if not api_key and not (provider == 'openai' and openai_base_url):
return _reply({'success': False, 'models': [], 'message': 'API key required'})
return jsonify({'success': False, 'models': [], 'message': 'API key required'})
from ai_providers import get_provider
ai_provider = get_provider(
@@ -706,7 +532,7 @@ def get_provider_models():
)
if not ai_provider:
return _reply({'success': False, 'models': [], 'message': f'Unknown provider: {provider}'})
return jsonify({'success': False, 'models': [], 'message': f'Unknown provider: {provider}'})
# Get all models from provider API
api_models = ai_provider.list_models()
@@ -725,13 +551,13 @@ def get_provider_models():
# so "gpt-4o-mini" as a fallback would be misleading).
if verified_models and not is_openai_compat:
models = sorted(verified_models)
return _reply({
return jsonify({
'success': True,
'models': models,
'recommended': recommended or models[0],
'message': f'{len(models)} verified models (API unavailable)'
})
return _reply({
return jsonify({
'success': False,
'models': [],
'message': 'Could not retrieve models. Check your API key and endpoint URL.'
@@ -741,7 +567,7 @@ def get_provider_models():
# Custom OpenAI-compatible endpoint: surface every model the
# endpoint reports. No verified-list intersection.
models = sorted(api_models)
return _reply({
return jsonify({
'success': True,
'models': models,
'recommended': models[0] if models else '',
@@ -769,7 +595,7 @@ def get_provider_models():
# No verified list for this provider, return all from API
models = sorted(api_models)
return _reply({
return jsonify({
'success': True,
'models': models,
'recommended': recommended if recommended in models else (models[0] if models else ''),
@@ -777,20 +603,11 @@ def get_provider_models():
})
except Exception as e:
# Outside the _reply closure — build the payload manually so the
# catch-all still carries catalog_meta when a refresh was
# attempted before the crash.
payload = {
return jsonify({
'success': False,
'models': [],
'message': f'Error: {str(e)}',
}
try:
if catalog_meta is not None:
payload['catalog_meta'] = catalog_meta
except UnboundLocalError:
pass
return jsonify(payload)
'message': f'Error: {str(e)}'
})
@notification_bp.route('/api/notifications/test-ai', methods=['POST'])