mirror of
https://github.com/MacRimi/ProxMenux.git
synced 2026-09-14 18:56:52 +00:00
drop live AI catalog refresh + GH Action
This commit is contained in:
@@ -1,16 +0,0 @@
|
||||
# ProxMenux AI-model verifier — API keys
|
||||
# Copy this file to .env (gitignored) and fill only the ones you have.
|
||||
|
||||
OPENAI_API_KEY=
|
||||
# Optional: for LiteLLM/MLX/LM Studio/vLLM/LocalAI/Ollama-proxy testing,
|
||||
# point OPENAI_API_KEY to any non-empty placeholder and set OPENAI_BASE_URL
|
||||
# to the endpoint's root (without /v1 — the tool appends it).
|
||||
# OPENAI_BASE_URL=http://localhost:4000
|
||||
|
||||
GROQ_API_KEY=
|
||||
|
||||
GEMINI_API_KEY=
|
||||
|
||||
ANTHROPIC_API_KEY=
|
||||
|
||||
OPENROUTER_API_KEY=
|
||||
@@ -1,37 +0,0 @@
|
||||
# AI models verifier (public copy)
|
||||
|
||||
Standalone verifier used by the daily GitHub Action to refresh
|
||||
`AppImage/config/verified_ai_models.json`.
|
||||
|
||||
The code lives here so the Action can execute it. API keys are read from
|
||||
GitHub Secrets at run time and never written to disk.
|
||||
|
||||
Local dev runs (interactive verifier over your own keys) can keep using
|
||||
the private copy — `verify.py` is identical.
|
||||
|
||||
## What the Action does
|
||||
|
||||
Each run:
|
||||
|
||||
1. Loads keys from Secrets into environment variables.
|
||||
2. Runs `verify.py --json-out /tmp/report.json` against every provider that
|
||||
has a key set.
|
||||
3. Rewrites `AppImage/config/verified_ai_models.json` with the passing
|
||||
models, sorted with the recommended one first per provider.
|
||||
4. Bumps the `_updated` field to the current date.
|
||||
5. If the file changed, commits directly to `main` as a bot commit.
|
||||
|
||||
## Adding provider keys
|
||||
|
||||
- Repository → Settings → Secrets and variables → Actions.
|
||||
- Add each key with the exact name expected by `verify.py`:
|
||||
`OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `GROQ_API_KEY`, `GEMINI_API_KEY`,
|
||||
`OPENROUTER_API_KEY`.
|
||||
- Any provider without a key is silently skipped — the Action logs a
|
||||
warning and continues with the rest.
|
||||
|
||||
## Running the Action on demand
|
||||
|
||||
The workflow accepts `workflow_dispatch`, so you can trigger a refresh
|
||||
manually from the Actions tab. Useful when a new model has just been
|
||||
released upstream and you don't want to wait for the daily cron.
|
||||
@@ -1,143 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Apply a verifier report to ``AppImage/config/verified_ai_models.json``.
|
||||
|
||||
Reads the machine-readable report emitted by ``verify.py --json-out`` and
|
||||
merges the passing models into the on-disk catalog. Preserves the
|
||||
maintainer's editorial curation across three axes:
|
||||
|
||||
* ``_exclude``: per-provider list of model IDs (exact match) that must
|
||||
never appear in the surfaced ``models`` list even when the verifier
|
||||
passes them. Meant for models that respond correctly to the technical
|
||||
test but are the wrong fit for notification translation — Arabic-only
|
||||
bases, Chinese-first fine-tunes, safety-classifier variants,
|
||||
agentic-only endpoints, legacy dated snapshots, etc.
|
||||
* ``recommended``: if the current recommendation is still in the
|
||||
passing (and non-excluded) set, it is preserved. Only when the
|
||||
previous recommendation disappears (deprecated upstream, or newly
|
||||
excluded) is a fallback chosen — the fastest passing model.
|
||||
* ``_note`` / ``_deprecated``: never touched. Those are maintainer
|
||||
annotations that outlive any single verifier run.
|
||||
|
||||
Fail-safe rules:
|
||||
* Providers absent from the report (no API key configured in the
|
||||
Action for that run) are left untouched.
|
||||
* Providers whose report carries an error are left untouched.
|
||||
* If the ``_exclude`` filter drops every passing model, the block is
|
||||
left untouched — an empty models list would silently kill the
|
||||
provider in the UI; keeping the previous list is more forgiving
|
||||
than shipping "nothing works".
|
||||
* ``_updated`` bumps to today's date only when the merge actually
|
||||
changed something. A no-op run leaves the file byte-identical.
|
||||
|
||||
Exits 0 when the file is unchanged, 10 when it was updated. The
|
||||
workflow uses that exit code to decide whether to commit.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import datetime as dt
|
||||
import fnmatch
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def _load_json(path: Path) -> dict:
|
||||
with open(path, "r", encoding="utf-8") as fh:
|
||||
return json.load(fh)
|
||||
|
||||
|
||||
def _save_json(path: Path, data: dict) -> None:
|
||||
tmp = path.with_suffix(path.suffix + ".tmp")
|
||||
with open(tmp, "w", encoding="utf-8") as fh:
|
||||
json.dump(data, fh, indent=2, ensure_ascii=False)
|
||||
fh.write("\n")
|
||||
tmp.replace(path)
|
||||
|
||||
|
||||
def _is_excluded(model: str, patterns: list[str]) -> bool:
|
||||
"""Match a model against the ``_exclude`` list. Supports exact
|
||||
matches and shell-style globs (``gpt-4o-*``, ``*-2024-*``, ...) so
|
||||
a provider that periodically publishes dated snapshots can be
|
||||
covered by a single pattern instead of one entry per date."""
|
||||
for pat in patterns:
|
||||
if pat == model or fnmatch.fnmatchcase(model, pat):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _passing_models(provider_report: dict, exclude: list[str]) -> list[str]:
|
||||
"""Passing models minus the editorial exclusion list, fastest first."""
|
||||
passing = [
|
||||
r for r in provider_report.get("results", [])
|
||||
if r.get("verdict") == "pass" and not _is_excluded(r.get("model", ""), exclude)
|
||||
]
|
||||
passing.sort(key=lambda r: r.get("latency_s", 999))
|
||||
return [r["model"] for r in passing]
|
||||
|
||||
|
||||
def apply_report(report_path: Path, catalog_path: Path, today: str) -> bool:
|
||||
"""Rewrite the catalog from the report. Returns True if it changed."""
|
||||
report = _load_json(report_path)
|
||||
catalog = _load_json(catalog_path) if catalog_path.exists() else {}
|
||||
|
||||
changed = False
|
||||
for provider_report in report:
|
||||
name = provider_report.get("provider")
|
||||
if not name:
|
||||
continue
|
||||
if provider_report.get("error"):
|
||||
print(f"[{name}] skipped — verifier reported error: {provider_report['error']}",
|
||||
file=sys.stderr)
|
||||
continue
|
||||
|
||||
block = catalog.setdefault(name, {})
|
||||
exclude = list(block.get("_exclude", []))
|
||||
passing = _passing_models(provider_report, exclude)
|
||||
|
||||
if not passing:
|
||||
# Either the verifier returned no passes for this provider,
|
||||
# or every pass got filtered by _exclude. Both cases mean
|
||||
# "no signal we can trust to overwrite the curated list";
|
||||
# leaving the block alone is safer than blanking it.
|
||||
print(f"[{name}] skipped — no passing models after exclude filter",
|
||||
file=sys.stderr)
|
||||
continue
|
||||
|
||||
prev_models = list(block.get("models", []))
|
||||
prev_recommended = block.get("recommended", "")
|
||||
# Preserve the maintainer's choice of recommended when it is
|
||||
# still valid. Only fall back to fastest when the previous
|
||||
# value disappeared from the passing set.
|
||||
recommended = prev_recommended if prev_recommended in passing else passing[0]
|
||||
|
||||
if sorted(prev_models) != sorted(passing) or prev_recommended != recommended:
|
||||
block["models"] = passing
|
||||
block["recommended"] = recommended
|
||||
changed = True
|
||||
print(f"[{name}] updated — {len(passing)} models, recommended={recommended}")
|
||||
else:
|
||||
print(f"[{name}] unchanged — {len(passing)} models")
|
||||
|
||||
if changed:
|
||||
catalog["_updated"] = today
|
||||
_save_json(catalog_path, catalog)
|
||||
return changed
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("--report", required=True, help="verify.py --json-out path")
|
||||
ap.add_argument("--catalog", required=True,
|
||||
help="AppImage/config/verified_ai_models.json path")
|
||||
ap.add_argument("--today", default=None,
|
||||
help="Override the date written into _updated (YYYY-MM-DD).")
|
||||
args = ap.parse_args()
|
||||
|
||||
today = args.today or dt.datetime.utcnow().strftime("%Y-%m-%d")
|
||||
changed = apply_report(Path(args.report), Path(args.catalog), today)
|
||||
return 10 if changed else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,50 +0,0 @@
|
||||
"""Standardized test prompt for ProxMenux AI-model verification.
|
||||
|
||||
Mirrors the real AI-enrichment use case: take a raw Proxmox system
|
||||
notification (English, with technical identifiers), translate it into
|
||||
Spanish, explain in plain terms, and suggest one concrete action. It is
|
||||
intentionally simple — if a model can't do this, it won't do the real
|
||||
thing either. Models that pass this test are fine for inclusion in
|
||||
verified_ai_models.json.
|
||||
"""
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"You are a Proxmox system-notification assistant. "
|
||||
"When given a raw notification from a Proxmox host, you: "
|
||||
"(1) translate it into Spanish, "
|
||||
"(2) explain in 2-3 sentences what the user is seeing and the likely cause, "
|
||||
"(3) suggest ONE concrete next action. "
|
||||
"Keep technical identifiers (device paths like /dev/sdd, SMART keywords, "
|
||||
"ata port numbers, BDFs) in their original form. "
|
||||
"Respond only in Spanish. Stay under 200 tokens total."
|
||||
)
|
||||
|
||||
# Realistic ProxMenux notification payload: multi-line body with
|
||||
# SMART/ATA vocabulary and a frequency hint — the exact shape the real
|
||||
# pipeline emits.
|
||||
USER_MESSAGE = (
|
||||
"Event: disk_io_error\n"
|
||||
"Severity: CRITICAL\n"
|
||||
"Host: pve-constructor\n"
|
||||
"Device: /dev/sdd\n"
|
||||
"SMART status: PASSED\n"
|
||||
"Summary: 3 I/O event(s) in 5 minutes, disk passed SMART short test\n"
|
||||
"Sample kernel line: ata4.00: exception Emask 0x0 SAct 0x804000 SErr 0x0 action 0x6\n"
|
||||
"Frequency: 3 occurrences in 24h, first seen 6h ago"
|
||||
)
|
||||
|
||||
# Common Spanish stopwords. A response missing ALL of these is almost
|
||||
# certainly not Spanish (or empty/truncated). Cheap heuristic, good
|
||||
# enough for a coarse pass/fail.
|
||||
REQUIRED_SPANISH_HINTS = [
|
||||
" el ", " la ", " los ", " las ", " un ", " una ",
|
||||
" de ", " del ", " en ", " con ", " que ", " para ",
|
||||
" es ", " se ", " ha ", " por ", " y ",
|
||||
]
|
||||
|
||||
# Domain keywords — at least one must appear to confirm the model
|
||||
# actually engaged with the notification instead of replying generically.
|
||||
DOMAIN_HINTS = [
|
||||
"disco", "sdd", "smart", "ata", "i/o", "e/s", "error",
|
||||
"kernel", "proxmox",
|
||||
]
|
||||
@@ -1,263 +0,0 @@
|
||||
"""Self-contained API wrappers for AI-model verification.
|
||||
|
||||
Kept independent from the ProxMenux AppImage's ai_providers module so
|
||||
this tool can live in a private repo with no import coupling to the
|
||||
public project. Uses only the Python standard library.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from typing import List, Optional
|
||||
|
||||
|
||||
class ProviderError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class Provider:
|
||||
"""Base class. Subclasses implement list_models() and generate()."""
|
||||
name = "base"
|
||||
|
||||
def __init__(self, api_key: str, base_url: Optional[str] = None):
|
||||
self.api_key = api_key
|
||||
self.base_url = base_url
|
||||
|
||||
def list_models(self) -> List[str]:
|
||||
raise NotImplementedError
|
||||
|
||||
def generate(self, model: str, system: str, user: str,
|
||||
max_tokens: int = 250, timeout: int = 30) -> str:
|
||||
raise NotImplementedError
|
||||
|
||||
# ── HTTP helpers ────────────────────────────────────────────
|
||||
|
||||
# Cloudflare in front of api.groq.com (and probably other providers
|
||||
# over time) returns 403 "error code: 1010" for the default
|
||||
# `Python-urllib/3.x` User-Agent — the "browser signature ban" rule.
|
||||
# A plain identifier is enough to get through; we're not spoofing a
|
||||
# browser, just avoiding a naive UA fingerprint match.
|
||||
_USER_AGENT = "ProxMenux-AI-Verifier/1.0"
|
||||
|
||||
def _post_json(self, url: str, payload: dict, headers: dict,
|
||||
timeout: int = 30) -> dict:
|
||||
data = json.dumps(payload).encode("utf-8")
|
||||
headers = {**headers, "User-Agent": self._USER_AGENT}
|
||||
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
return json.loads(resp.read().decode("utf-8"))
|
||||
|
||||
def _get_json(self, url: str, headers: dict, timeout: int = 30) -> dict:
|
||||
headers = {**headers, "User-Agent": self._USER_AGENT}
|
||||
req = urllib.request.Request(url, headers=headers, method="GET")
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
return json.loads(resp.read().decode("utf-8"))
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════
|
||||
# OpenAI-compatible family (OpenAI, Groq, OpenRouter, LiteLLM, LM
|
||||
# Studio, vLLM, LocalAI, etc.). They all speak the same endpoints.
|
||||
# ══════════════════════════════════════════════════════════════════
|
||||
|
||||
class OpenAICompatProvider(Provider):
|
||||
name = "openai-compat"
|
||||
default_base = "https://api.openai.com"
|
||||
models_path = "/v1/models"
|
||||
chat_path = "/v1/chat/completions"
|
||||
|
||||
def _base(self) -> str:
|
||||
return (self.base_url or self.default_base).rstrip("/")
|
||||
|
||||
@staticmethod
|
||||
def _is_reasoning_model(model: str) -> bool:
|
||||
"""True for OpenAI reasoning models (o-series + non-chat gpt-5+).
|
||||
|
||||
Must be kept in sync with the matching helper in ProxMenux's
|
||||
openai_provider.py — same rule, same consequence:
|
||||
- send max_completion_tokens instead of max_tokens
|
||||
- omit temperature (default is the only accepted value).
|
||||
"""
|
||||
m = model.lower()
|
||||
if len(m) >= 2 and m[0] == "o" and m[1].isdigit():
|
||||
return True
|
||||
if m.startswith("gpt-5") and "-chat" not in m:
|
||||
return True
|
||||
return False
|
||||
|
||||
def list_models(self) -> List[str]:
|
||||
url = f"{self._base()}{self.models_path}"
|
||||
headers = {"Authorization": f"Bearer {self.api_key}"}
|
||||
data = self._get_json(url, headers)
|
||||
return [m.get("id", "") for m in data.get("data", []) if m.get("id")]
|
||||
|
||||
def generate(self, model: str, system: str, user: str,
|
||||
max_tokens: int = 250, timeout: int = 30) -> str:
|
||||
url = f"{self._base()}{self.chat_path}"
|
||||
headers = {
|
||||
"Authorization": f"Bearer {self.api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [
|
||||
{"role": "system", "content": system},
|
||||
{"role": "user", "content": user},
|
||||
],
|
||||
}
|
||||
if self._is_reasoning_model(model):
|
||||
# Reasoning models spend budget on internal reasoning by
|
||||
# default, which yields empty replies at small max_tokens.
|
||||
# reasoning_effort=minimal keeps that overhead low so the
|
||||
# whole budget reaches the user, aligned with the short
|
||||
# translate+explain task ProxMenux uses. Mirror this in
|
||||
# ProxMenux's openai_provider.py.
|
||||
payload["max_completion_tokens"] = max_tokens
|
||||
payload["reasoning_effort"] = "minimal"
|
||||
else:
|
||||
payload["max_tokens"] = max_tokens
|
||||
payload["temperature"] = 0.3
|
||||
data = self._post_json(url, payload, headers, timeout)
|
||||
try:
|
||||
return data["choices"][0]["message"]["content"].strip()
|
||||
except (KeyError, IndexError) as exc:
|
||||
raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}")
|
||||
|
||||
|
||||
class OpenAIProvider(OpenAICompatProvider):
|
||||
name = "openai"
|
||||
default_base = "https://api.openai.com"
|
||||
|
||||
|
||||
class GroqProvider(OpenAICompatProvider):
|
||||
name = "groq"
|
||||
default_base = "https://api.groq.com/openai"
|
||||
|
||||
|
||||
class OpenRouterProvider(OpenAICompatProvider):
|
||||
name = "openrouter"
|
||||
default_base = "https://openrouter.ai/api"
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════
|
||||
# Gemini — different endpoint shape, keyed via ?key=... query param.
|
||||
# ══════════════════════════════════════════════════════════════════
|
||||
|
||||
class GeminiProvider(Provider):
|
||||
name = "gemini"
|
||||
default_base = "https://generativelanguage.googleapis.com/v1beta"
|
||||
|
||||
@staticmethod
|
||||
def _has_thinking_mode(model: str) -> bool:
|
||||
"""True for Gemini variants that enable "thinking" by default.
|
||||
|
||||
Kept in sync with ProxMenux's gemini_provider.py. 2.5+ pro/flash
|
||||
and 3.x pro/flash consume output tokens on reasoning, which
|
||||
yields empty replies when max_tokens is small. We pass
|
||||
thinkingBudget=0 to disable thinking so the short translate+
|
||||
explain test sees actual text. Lite variants don't have thinking
|
||||
enabled and are not flagged here.
|
||||
"""
|
||||
m = model.lower()
|
||||
if "lite" in m:
|
||||
return False
|
||||
return m.startswith("gemini-2.5") or m.startswith("gemini-3")
|
||||
|
||||
def list_models(self) -> List[str]:
|
||||
url = f"{self.default_base}/models?key={self.api_key}"
|
||||
data = self._get_json(url, {})
|
||||
names = []
|
||||
for m in data.get("models", []):
|
||||
raw = m.get("name", "")
|
||||
if raw.startswith("models/"):
|
||||
raw = raw[len("models/"):]
|
||||
# Only keep text-generation capable models.
|
||||
if "generateContent" in m.get("supportedGenerationMethods", []):
|
||||
names.append(raw)
|
||||
return names
|
||||
|
||||
def generate(self, model: str, system: str, user: str,
|
||||
max_tokens: int = 250, timeout: int = 30) -> str:
|
||||
url = f"{self.default_base}/models/{model}:generateContent?key={self.api_key}"
|
||||
gen_config = {
|
||||
"maxOutputTokens": max_tokens,
|
||||
"temperature": 0.3,
|
||||
}
|
||||
if self._has_thinking_mode(model):
|
||||
gen_config["thinkingConfig"] = {"thinkingBudget": 0}
|
||||
payload = {
|
||||
"system_instruction": {"parts": [{"text": system}]},
|
||||
"contents": [{"parts": [{"text": user}]}],
|
||||
"generationConfig": gen_config,
|
||||
}
|
||||
data = self._post_json(url, payload, {"Content-Type": "application/json"}, timeout)
|
||||
try:
|
||||
return data["candidates"][0]["content"]["parts"][0]["text"].strip()
|
||||
except (KeyError, IndexError) as exc:
|
||||
raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}")
|
||||
|
||||
|
||||
# ══════════════════════════════════════════════════════════════════
|
||||
# Anthropic — own schema and headers.
|
||||
# ══════════════════════════════════════════════════════════════════
|
||||
|
||||
class AnthropicProvider(Provider):
|
||||
name = "anthropic"
|
||||
default_base = "https://api.anthropic.com"
|
||||
version = "2023-06-01"
|
||||
|
||||
def list_models(self) -> List[str]:
|
||||
url = f"{self.default_base}/v1/models"
|
||||
headers = {
|
||||
"x-api-key": self.api_key,
|
||||
"anthropic-version": self.version,
|
||||
}
|
||||
try:
|
||||
data = self._get_json(url, headers)
|
||||
return [m.get("id", "") for m in data.get("data", []) if m.get("id")]
|
||||
except urllib.error.HTTPError:
|
||||
# Older keys don't have the endpoint; caller decides what to do.
|
||||
return []
|
||||
|
||||
def generate(self, model: str, system: str, user: str,
|
||||
max_tokens: int = 250, timeout: int = 30) -> str:
|
||||
url = f"{self.default_base}/v1/messages"
|
||||
headers = {
|
||||
"x-api-key": self.api_key,
|
||||
"anthropic-version": self.version,
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
# ProxMenux's real anthropic_provider.py does NOT send `temperature`,
|
||||
# and Anthropic's newest generation (Claude Sonnet 5, Opus 4.7 /
|
||||
# 4.8, Fable 5) now rejects it with:
|
||||
# invalid_request_error: `temperature` is deprecated for this model.
|
||||
# Omitting it here aligns the verifier with production behaviour so
|
||||
# the newest-gen models can be reached during verification.
|
||||
payload = {
|
||||
"model": model,
|
||||
"max_tokens": max_tokens,
|
||||
"system": system,
|
||||
"messages": [{"role": "user", "content": user}],
|
||||
}
|
||||
data = self._post_json(url, payload, headers, timeout)
|
||||
try:
|
||||
return data["content"][0]["text"].strip()
|
||||
except (KeyError, IndexError) as exc:
|
||||
raise ProviderError(f"unexpected response: {exc} // {str(data)[:200]}")
|
||||
|
||||
|
||||
PROVIDERS = {
|
||||
"openai": OpenAIProvider,
|
||||
"groq": GroqProvider,
|
||||
"gemini": GeminiProvider,
|
||||
"anthropic": AnthropicProvider,
|
||||
"openrouter": OpenRouterProvider,
|
||||
}
|
||||
|
||||
|
||||
def make_provider(name: str, api_key: str,
|
||||
base_url: Optional[str] = None) -> Provider:
|
||||
cls = PROVIDERS.get(name)
|
||||
if not cls:
|
||||
raise ValueError(f"unknown provider: {name}")
|
||||
return cls(api_key, base_url=base_url)
|
||||
@@ -1,232 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""ProxMenux AI-model verifier.
|
||||
|
||||
Runs a standardized translate+explain test against every model each
|
||||
provider currently advertises, and emits a per-model verdict so the
|
||||
verified_ai_models.json list can be refreshed with confidence.
|
||||
|
||||
Not packaged with the AppImage — keep this in a private repo alongside
|
||||
the API keys.
|
||||
|
||||
Usage:
|
||||
cp .env.example .env # fill in the API keys you have
|
||||
python3 verify.py # test all providers with keys
|
||||
python3 verify.py --provider groq # just one
|
||||
python3 verify.py --provider openai --limit 5 # only first 5 models
|
||||
python3 verify.py --json-out report.json # machine-readable output too
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
from prompts import DOMAIN_HINTS, REQUIRED_SPANISH_HINTS, SYSTEM_PROMPT, USER_MESSAGE
|
||||
from providers import PROVIDERS, make_provider
|
||||
|
||||
|
||||
# Non-chat model name patterns. Skipping these saves test time and keeps
|
||||
# the report focused on models that could actually serve notifications.
|
||||
SKIP_PATTERNS = (
|
||||
"embedding", "whisper", "tts", "dall-e", "dalle", "image",
|
||||
"realtime", "audio", "moderation", "search",
|
||||
"code-search", "text-similarity", "babbage", "davinci",
|
||||
"curie", "ada", "transcribe",
|
||||
)
|
||||
|
||||
|
||||
def load_env(env_path: Path) -> Dict[str, str]:
|
||||
"""Minimal .env loader (avoids a python-dotenv dependency)."""
|
||||
if not env_path.exists():
|
||||
return {}
|
||||
env: Dict[str, str] = {}
|
||||
for line in env_path.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#") or "=" not in line:
|
||||
continue
|
||||
k, v = line.split("=", 1)
|
||||
env[k.strip()] = v.strip().strip('"').strip("'")
|
||||
return env
|
||||
|
||||
|
||||
def should_skip_model(model: str) -> bool:
|
||||
m = model.lower()
|
||||
return any(p in m for p in SKIP_PATTERNS)
|
||||
|
||||
|
||||
def assess_response(text: str) -> Tuple[str, List[str]]:
|
||||
"""Classify the model output. Returns (verdict, reasons).
|
||||
|
||||
verdict is one of:
|
||||
- 'pass': Spanish, on-topic, reasonable length
|
||||
- 'warn': responded but one heuristic failed (borderline)
|
||||
- 'fail': empty, wrong language, or off-topic
|
||||
"""
|
||||
reasons: List[str] = []
|
||||
if not text or len(text) < 30:
|
||||
return "fail", ["empty or too short response"]
|
||||
|
||||
text_low = " " + text.lower() + " "
|
||||
spanish_hits = sum(1 for h in REQUIRED_SPANISH_HINTS if h in text_low)
|
||||
domain_hits = sum(1 for h in DOMAIN_HINTS if h.lower() in text_low)
|
||||
|
||||
if spanish_hits < 3:
|
||||
reasons.append(f"not Spanish ({spanish_hits}/{len(REQUIRED_SPANISH_HINTS)} hints)")
|
||||
if domain_hits < 1:
|
||||
reasons.append("did not engage with the domain")
|
||||
if len(text) > 1500:
|
||||
reasons.append("response unusually long")
|
||||
|
||||
if not reasons:
|
||||
return "pass", []
|
||||
# Responded and engaged, but one signal missed → warn (keep in list
|
||||
# with a caveat; don't auto-include).
|
||||
if domain_hits >= 1 and spanish_hits >= 1:
|
||||
return "warn", reasons
|
||||
return "fail", reasons
|
||||
|
||||
|
||||
def run_model(provider, model: str, timeout: int) -> dict:
|
||||
t0 = time.time()
|
||||
try:
|
||||
out = provider.generate(
|
||||
model, SYSTEM_PROMPT, USER_MESSAGE,
|
||||
max_tokens=250, timeout=timeout,
|
||||
)
|
||||
latency = time.time() - t0
|
||||
verdict, reasons = assess_response(out)
|
||||
return {
|
||||
"model": model,
|
||||
"verdict": verdict,
|
||||
"latency_s": round(latency, 2),
|
||||
"reasons": reasons,
|
||||
"sample": (out[:140] if out else "").replace("\n", " "),
|
||||
"error": None,
|
||||
}
|
||||
except Exception as exc: # HTTPError, ProviderError, timeouts
|
||||
return {
|
||||
"model": model,
|
||||
"verdict": "fail",
|
||||
"latency_s": round(time.time() - t0, 2),
|
||||
"reasons": [],
|
||||
"sample": "",
|
||||
"error": str(exc)[:200],
|
||||
}
|
||||
|
||||
|
||||
def tag(verdict: str) -> str:
|
||||
return {"pass": "✓", "warn": "⚠", "fail": "✗"}.get(verdict, "?")
|
||||
|
||||
|
||||
def run_provider(name: str, api_key: str, base_url: Optional[str],
|
||||
timeout: int, limit: Optional[int]) -> dict:
|
||||
print(f"\n=== {name} ===")
|
||||
try:
|
||||
provider = make_provider(name, api_key, base_url=base_url)
|
||||
models = provider.list_models()
|
||||
except Exception as exc:
|
||||
print(f" list_models() failed: {exc}")
|
||||
return {"provider": name, "error": str(exc), "results": []}
|
||||
|
||||
models = [m for m in models if not should_skip_model(m)]
|
||||
if limit:
|
||||
models = models[:limit]
|
||||
if not models:
|
||||
print(" (no eligible models)")
|
||||
return {"provider": name, "error": None, "results": []}
|
||||
|
||||
print(f" discovered {len(models)} model(s)")
|
||||
results: List[dict] = []
|
||||
for m in models:
|
||||
r = run_model(provider, m, timeout)
|
||||
results.append(r)
|
||||
latency = f"{r['latency_s']}s"
|
||||
if r["error"]:
|
||||
suffix = f" — {r['error']}"
|
||||
elif r["reasons"]:
|
||||
suffix = f" — {'; '.join(r['reasons'])}"
|
||||
else:
|
||||
suffix = ""
|
||||
print(f" {tag(r['verdict'])} {m:<50} {latency:>6}{suffix}")
|
||||
|
||||
return {"provider": name, "error": None, "results": results}
|
||||
|
||||
|
||||
def summarize(all_results: List[dict]) -> None:
|
||||
print("\n" + "=" * 64)
|
||||
print("Suggested verified_ai_models.json entries (passing models only)")
|
||||
print("=" * 64)
|
||||
any_output = False
|
||||
for pr in all_results:
|
||||
if pr.get("error"):
|
||||
continue
|
||||
passed = [r for r in pr["results"] if r["verdict"] == "pass"]
|
||||
if not passed:
|
||||
continue
|
||||
any_output = True
|
||||
passed_sorted = sorted(passed, key=lambda x: x["latency_s"])
|
||||
print(f'\n "{pr["provider"]}": {{')
|
||||
print(' "models": [')
|
||||
for r in passed_sorted:
|
||||
print(f' "{r["model"]}",')
|
||||
print(" ],")
|
||||
print(f' "recommended": "{passed_sorted[0]["model"]}"')
|
||||
print(" },")
|
||||
|
||||
warn_total = sum(
|
||||
1 for pr in all_results for r in pr["results"]
|
||||
if r["verdict"] == "warn"
|
||||
)
|
||||
if warn_total:
|
||||
print(f"\n Note: {warn_total} model(s) came back as ⚠ (warn) — review those manually.")
|
||||
if not any_output:
|
||||
print("\n (no models passed; check keys and network)")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(
|
||||
description="Verify AI models for ProxMenux notification enrichment."
|
||||
)
|
||||
ap.add_argument("--env", default=".env", help=".env file path")
|
||||
ap.add_argument("--provider", action="append", default=[],
|
||||
help="run a specific provider (repeat to add more)")
|
||||
ap.add_argument("--timeout", type=int, default=30,
|
||||
help="seconds per request (default: 30)")
|
||||
ap.add_argument("--limit", type=int, default=None,
|
||||
help="test at most N models per provider (debug)")
|
||||
ap.add_argument("--json-out", default=None,
|
||||
help="write machine-readable report to this path")
|
||||
args = ap.parse_args()
|
||||
|
||||
env = {**os.environ, **load_env(Path(args.env))}
|
||||
|
||||
provider_list = args.provider or list(PROVIDERS.keys())
|
||||
tested: List[dict] = []
|
||||
for name in provider_list:
|
||||
if name not in PROVIDERS:
|
||||
print(f"unknown provider: {name}", file=sys.stderr)
|
||||
continue
|
||||
key_var = f"{name.upper()}_API_KEY"
|
||||
url_var = f"{name.upper()}_BASE_URL"
|
||||
api_key = env.get(key_var, "")
|
||||
base_url = env.get(url_var) or None
|
||||
if not api_key:
|
||||
print(f"\n=== {name} ===\n skipped — {key_var} not set")
|
||||
continue
|
||||
tested.append(run_provider(name, api_key, base_url, args.timeout, args.limit))
|
||||
|
||||
summarize(tested)
|
||||
|
||||
if args.json_out:
|
||||
Path(args.json_out).write_text(json.dumps(tested, indent=2))
|
||||
print(f"\nReport written to {args.json_out}")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,103 +0,0 @@
|
||||
name: Verify AI models catalog
|
||||
|
||||
# Runs the AI-model verifier and commits any changes to
|
||||
# AppImage/config/verified_ai_models.json on the same branch the run
|
||||
# was launched from.
|
||||
#
|
||||
# GitHub only fires `on: schedule` from the default branch, so the
|
||||
# daily cron always runs against main and keeps stable users fresh.
|
||||
# When a beta cycle needs its own refresh on develop, use the
|
||||
# "Run workflow" button on the Actions tab and pick develop from the
|
||||
# branch selector — the same YAML then checks out develop, runs the
|
||||
# verifier and commits back to develop. Cross-branch pushes never
|
||||
# happen: each run only touches the branch it started on.
|
||||
#
|
||||
# The verifier code lives at .github/scripts/ai-models-verifier/ and
|
||||
# reads API keys from repository Secrets. Any provider without a key
|
||||
# is skipped silently — the workflow keeps going with the rest.
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: '0 4 * * *' # 04:00 UTC every day — cron always fires from main
|
||||
workflow_dispatch: # manual trigger — branch is picked in the UI
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
concurrency:
|
||||
# Keyed by branch so a manual develop run does not collide with the
|
||||
# scheduled main run — each branch gets its own serialisation lane.
|
||||
group: verify-ai-models-${{ github.ref_name }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
verify:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Check out the branch this run belongs to
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
# `github.ref_name` resolves to main for the cron and to the
|
||||
# branch selected in the dispatch UI otherwise. The same
|
||||
# value is used again below when we push, so every run is
|
||||
# symmetric: checkout X → refresh → push X.
|
||||
ref: ${{ github.ref_name }}
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Run verifier
|
||||
working-directory: .github/scripts/ai-models-verifier
|
||||
env:
|
||||
# API keys — each is optional. verify.py silently skips any
|
||||
# provider whose *_API_KEY env var is empty, so the workflow
|
||||
# runs even when only a subset of keys is configured.
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
GROQ_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
|
||||
# Optional base URLs for custom-endpoint providers.
|
||||
OPENAI_BASE_URL: ${{ secrets.OPENAI_BASE_URL }}
|
||||
run: |
|
||||
python3 verify.py --json-out /tmp/report.json || true
|
||||
if [ ! -s /tmp/report.json ]; then
|
||||
echo "Verifier produced no report — bailing"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Apply report to catalog
|
||||
id: apply
|
||||
working-directory: .
|
||||
run: |
|
||||
set +e
|
||||
python3 .github/scripts/ai-models-verifier/apply.py \
|
||||
--report /tmp/report.json \
|
||||
--catalog AppImage/config/verified_ai_models.json
|
||||
code=$?
|
||||
set -e
|
||||
case "$code" in
|
||||
0) echo "changed=false" >> "$GITHUB_OUTPUT" ;;
|
||||
10) echo "changed=true" >> "$GITHUB_OUTPUT" ;;
|
||||
*) echo "apply.py exited with $code"; exit "$code" ;;
|
||||
esac
|
||||
|
||||
- name: Commit and push back to the same branch
|
||||
if: steps.apply.outputs.changed == 'true'
|
||||
run: |
|
||||
git config user.name "proxmenux-bot"
|
||||
git config user.email "proxmenux-bot@users.noreply.github.com"
|
||||
git add AppImage/config/verified_ai_models.json
|
||||
git commit -m "chore(ai-models): daily catalog refresh"
|
||||
# Push to the branch this run started on — same ref used at
|
||||
# checkout above, so the operation is symmetric regardless of
|
||||
# whether cron (main) or dispatch (any branch) triggered it.
|
||||
git push origin HEAD:${{ github.ref_name }}
|
||||
|
||||
- name: Report unchanged
|
||||
if: steps.apply.outputs.changed != 'true'
|
||||
run: echo "Catalog already up to date — nothing to commit."
|
||||
Reference in New Issue
Block a user