b805c294eb
Backend: FastAPI mit /api/health + /api/models (read-only aus llama-swap config, Logik aus v1 portiert). Frontend: Vite + React + Tailwind v4 + shadcn-Style App-Shell mit Cmd+K, Dark-default, PWA-Manifest; Modelle-View live aus /api/models. Deploy-Geruest (systemd-Unit :9001, build/deploy- Skripte, NOPASSWD-sudoers, kein Klartext-Passwort). Verifiziert: Backend-Endpunkte + Frontend-Build + gerenderte Shell. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
95 lines
3.0 KiB
Python
95 lines
3.0 KiB
Python
"""
|
|
Engine-Service: liest die llama-swap config.yaml (read-only in Phase 0) und
|
|
spricht die llama-swap-API (/v1/models, /running). Schreiblogik (Modelle
|
|
installieren, Gruppen/Routing verwalten) kommt in Phase 1.
|
|
|
|
Logik portiert aus Mission Control v1 (llamaswap.py + routers/models.py),
|
|
auf das Nötigste reduziert.
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
|
|
import httpx
|
|
|
|
from config import CONFIG_PATH, LLAMA_SWAP_URL, yaml
|
|
|
|
# Kanonische Rollen (vereinheitlicht ggü. v1: kein manager/reviewer mehr).
|
|
# Eine Quelle der Wahrheit — Capability-Erkennung läuft separat über model_caps (Phase 1).
|
|
ROLE_IDS = {"vision", "coder", "reasoning", "agent", "scout"}
|
|
|
|
_CTX_RE = re.compile(r"-(?:c|-ctx-size)\s+(\d+)")
|
|
_PATH_RE = re.compile(r"-(?:m|-model)\s+([^\s]+)")
|
|
_QUANT_RE = re.compile(r"(Q\d_[A-Z0-9_]+|IQ\d_[A-Z0-9_]+|fp16|bf16)\.gguf", re.IGNORECASE)
|
|
|
|
|
|
def read_config() -> dict:
|
|
"""llama-swap config.yaml laden. Existiert sie nicht (z.B. lokaler Dev-PC
|
|
ohne Engine), wird ein leeres Modell-Set zurückgegeben statt zu werfen."""
|
|
if not CONFIG_PATH.exists():
|
|
return {"models": {}}
|
|
with CONFIG_PATH.open("r", encoding="utf-8") as f:
|
|
data = yaml.load(f) or {}
|
|
if not data.get("models"):
|
|
data["models"] = {}
|
|
return data
|
|
|
|
|
|
def _parse_model(name: str, spec: dict) -> dict:
|
|
"""Ein config.yaml-Modell in ein flaches UI-Objekt übersetzen."""
|
|
spec = spec or {}
|
|
cmd = str(spec.get("cmd", "")).strip()
|
|
|
|
ctx = None
|
|
if (m := _CTX_RE.search(cmd)):
|
|
ctx = int(m.group(1))
|
|
|
|
path = ""
|
|
filename = ""
|
|
quant = ""
|
|
size_bytes = None
|
|
if (m := _PATH_RE.search(cmd)):
|
|
path = m.group(1).replace("'", "").replace('"', "")
|
|
filename = os.path.basename(path)
|
|
if os.path.exists(path):
|
|
size_bytes = os.path.getsize(path)
|
|
if (q := _QUANT_RE.search(path)):
|
|
quant = q.group(1).upper()
|
|
|
|
aliases = spec.get("aliases") or []
|
|
if isinstance(aliases, str):
|
|
aliases = [aliases]
|
|
aliases = [str(a) for a in aliases]
|
|
# Rolle = erster Alias; Legacy-Fallback: Key selbst ist eine Rolle.
|
|
role = aliases[0].lower() if aliases else (name.lower() if name.lower() in ROLE_IDS else None)
|
|
|
|
return {
|
|
"name": name,
|
|
"role": role,
|
|
"aliases": aliases,
|
|
"api_ids": [name] + aliases,
|
|
"ctx": ctx,
|
|
"ttl": spec.get("ttl"),
|
|
"cmd": cmd,
|
|
"filename": filename,
|
|
"quant": quant,
|
|
"size_bytes": size_bytes,
|
|
# "incomplete" = Eintrag ohne hinterlegtes Modell (-m), z.B. Platzhalter.
|
|
"incomplete": not path,
|
|
}
|
|
|
|
|
|
def list_models() -> list[dict]:
|
|
"""Alle in der config.yaml konfigurierten Modelle (read-only)."""
|
|
cfg = read_config()
|
|
return [_parse_model(name, spec) for name, spec in (cfg.get("models") or {}).items()]
|
|
|
|
|
|
def engine_reachable() -> bool:
|
|
"""Ist die llama-swap-API erreichbar?"""
|
|
try:
|
|
with httpx.Client(timeout=3.0) as c:
|
|
return c.get(f"{LLAMA_SWAP_URL}/v1/models").status_code == 200
|
|
except Exception:
|
|
return False
|