45ab32955a
- status(): Flag "incomplete" fuer Rollen ohne Modell (-m), z.B. provision.sh- Platzhalter. Wurden bisher irrefuehrend als "bereit / ? . 8K . ?" gezeigt. - Modelle-Tab: leere Rollen klar als "kein Modell hinterlegt / leer" + Aktion "Modell zuweisen" (-> Cookbook); nicht im Schnelltest-Dropdown waehlbar - register(): erkennt einen mmproj-Projektor im selben Ordner und ergaenzt --mmproj + --jinja automatisch (Vision braucht das in llama.cpp) Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
310 lines
11 KiB
Python
310 lines
11 KiB
Python
"""
|
|
Modelle-Router: Status, Download, Einpflegen, Unload, Schnelltest-Chat.
|
|
|
|
Bildet den Kern von Mission Control ab — alles, was direkt mit den llama-swap-
|
|
Modellen und ihrer config.yaml zu tun hat.
|
|
"""
|
|
|
|
from pathlib import Path
|
|
|
|
import httpx
|
|
from fastapi import APIRouter, Depends, HTTPException
|
|
from pydantic import BaseModel
|
|
from ruamel.yaml.scalarstring import LiteralScalarString
|
|
|
|
from auth import auth
|
|
from config import (CMD_TEMPLATE, CONFIG_PATH, DEFAULT_TTL, HF_DOWNLOAD_ENV, LLAMA_SWAP_URL,
|
|
MODELS_DIR, TOKEN, hf_bin)
|
|
from jobengine import JOBS, start_job, attach_download_progress
|
|
from routers.cookbook import hf_file_size
|
|
from llamaswap import _swap_get, read_config, write_config
|
|
from hw_math import extract_params_b, max_ctx_for, estimate_memory_gb
|
|
import re
|
|
import os
|
|
import shutil
|
|
import psutil
|
|
|
|
router = APIRouter(prefix="/api", dependencies=[Depends(auth)])
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Request-Modelle
|
|
# ---------------------------------------------------------------------------
|
|
class DownloadReq(BaseModel):
|
|
repo: str
|
|
file: str
|
|
subdir: str | None = None
|
|
hf_token: str | None = None
|
|
|
|
|
|
class RegisterReq(BaseModel):
|
|
alias: str
|
|
model_path: str
|
|
ctx: int = 8192
|
|
ttl: int | None = None
|
|
|
|
|
|
class ChatReq(BaseModel):
|
|
model: str
|
|
message: str
|
|
|
|
|
|
class UpdateReq(BaseModel):
|
|
alias: str
|
|
ctx: int
|
|
|
|
|
|
class DeleteReq(BaseModel):
|
|
alias: str
|
|
delete_files: bool = True
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Endpoints
|
|
# ---------------------------------------------------------------------------
|
|
@router.get("/status")
|
|
def status():
|
|
cfg = read_config()
|
|
try:
|
|
ram_gb = psutil.virtual_memory().total / (1024 ** 3)
|
|
except Exception: # noqa: BLE001
|
|
ram_gb = 0
|
|
configured = {}
|
|
for name, spec in (cfg.get("models") or {}).items():
|
|
spec = spec or {}
|
|
cmd = str(spec.get("cmd", "")).strip()
|
|
|
|
# Parse Meta
|
|
ctx = 8192
|
|
m_ctx = re.search(r'-(?:c|-ctx-size)\s+(\d+)', cmd)
|
|
if m_ctx: ctx = int(m_ctx.group(1))
|
|
|
|
size_bytes = None
|
|
quant = ""
|
|
filename = ""
|
|
m_path = re.search(r'-(?:m|-model)\s+([^\s]+)', cmd)
|
|
if m_path:
|
|
path = m_path.group(1).replace("'", "").replace('"', '')
|
|
if os.path.exists(path):
|
|
size_bytes = os.path.getsize(path)
|
|
filename = os.path.basename(path)
|
|
q_match = re.search(r'(Q\d_[A-Z0-9_]+|IQ\d_[A-Z0-9_]+|fp16|bf16)\.gguf', path, flags=re.IGNORECASE)
|
|
if q_match:
|
|
quant = q_match.group(1).upper()
|
|
|
|
caps = ["Text"]
|
|
if "coder" in name.lower() or (m_path and "code" in m_path.group(1).lower()):
|
|
caps = ["Code"]
|
|
if "--mmproj" in cmd:
|
|
caps.append("Bild")
|
|
|
|
# "incomplete" = Rolle existiert in der config, hat aber kein Modell (-m) hinterlegt
|
|
# (z.B. von provision.sh angelegter Platzhalter). Ehrlich kennzeichnen statt "bereit".
|
|
incomplete = not (m_path and m_path.group(1))
|
|
configured[name] = {
|
|
"name": name,
|
|
"ttl": spec.get("ttl", cfg.get("globalTTL", 0)),
|
|
"cmd": cmd,
|
|
"state": "idle",
|
|
"incomplete": incomplete,
|
|
"port": None,
|
|
"meta": {
|
|
"ctx": ctx,
|
|
"size_bytes": size_bytes,
|
|
"quant": quant,
|
|
"caps": caps,
|
|
"filename": filename,
|
|
"params_b": (_pb := extract_params_b(filename or name)),
|
|
"optimal_ctx": (_oc := (max_ctx_for(_pb, quant or "Q4_K_M", ram_gb) if ram_gb else None)),
|
|
"peak_ram_gb": round(estimate_memory_gb(_pb, quant or "Q4_K_M", ctx), 1),
|
|
"peak_ram_optimal_gb": (round(estimate_memory_gb(_pb, quant or "Q4_K_M", _oc), 1) if _oc else None),
|
|
}
|
|
}
|
|
swap_ok = True
|
|
try:
|
|
running = _swap_get("/running")
|
|
items = running.get("running", running) if isinstance(running, dict) else running
|
|
for item in items or []:
|
|
mid = item.get("model") or item.get("id") or item.get("name")
|
|
if mid in configured:
|
|
configured[mid]["state"] = item.get("state", "running")
|
|
configured[mid]["port"] = item.get("port")
|
|
elif mid:
|
|
configured[mid] = {
|
|
"name": mid, "ttl": None, "cmd": "",
|
|
"state": item.get("state", "running"), "port": item.get("port"),
|
|
}
|
|
except Exception: # noqa: BLE001
|
|
swap_ok = False
|
|
return {
|
|
"swap_ok": swap_ok,
|
|
"swap_url": LLAMA_SWAP_URL,
|
|
"config_path": str(CONFIG_PATH),
|
|
"models_dir": str(MODELS_DIR),
|
|
"secured": bool(TOKEN),
|
|
"models": list(configured.values()),
|
|
}
|
|
|
|
|
|
@router.post("/download")
|
|
def download(req: DownloadReq):
|
|
sub = req.subdir or req.repo.split("/")[-1]
|
|
target = MODELS_DIR / sub
|
|
target.mkdir(parents=True, exist_ok=True)
|
|
args = [hf_bin(), "download", req.repo, req.file, "--local-dir", str(target)]
|
|
env = dict(HF_DOWNLOAD_ENV)
|
|
if req.hf_token:
|
|
env["HF_TOKEN"] = req.hf_token
|
|
job_id = start_job(args, f"download {req.repo}/{req.file}", env=env)
|
|
JOBS[job_id]["result_path"] = str(target / req.file)
|
|
attach_download_progress(job_id, str(target), hf_file_size(req.repo, req.file))
|
|
return {"job_id": job_id, "expected_path": str(target / req.file)}
|
|
|
|
|
|
def _augment_vision(cmd: str, model_path: str) -> str:
|
|
"""Liegt im selben Ordner ein mmproj-Projektor, --mmproj + --jinja ergaenzen —
|
|
Vision-Modelle brauchen das in llama.cpp (siehe CLAUDE.md). No-op, wenn schon gesetzt
|
|
oder kein Projektor da."""
|
|
if "--mmproj" in cmd:
|
|
return cmd
|
|
try:
|
|
d = os.path.dirname(model_path)
|
|
for f in sorted(os.listdir(d)):
|
|
if "mmproj" in f.lower() and f.lower().endswith(".gguf"):
|
|
return cmd.rstrip() + f" --mmproj {os.path.join(d, f)} --jinja"
|
|
except OSError:
|
|
pass
|
|
return cmd
|
|
|
|
|
|
@router.post("/register")
|
|
def register(req: RegisterReq):
|
|
# Bewusst KEIN exists()-Check: beim frischen Download läuft der hf-Job noch, die Datei kommt
|
|
# erst gleich. Eintrag jetzt schon schreiben → llama-swap (-watch-config) lädt, sobald sie da ist.
|
|
cfg = read_config()
|
|
cmd = CMD_TEMPLATE.replace("{model}", req.model_path).replace("{ctx}", str(req.ctx))
|
|
cmd = _augment_vision(cmd, req.model_path)
|
|
cfg["models"][req.alias] = {
|
|
"cmd": LiteralScalarString(cmd + "\n"),
|
|
"ttl": req.ttl if req.ttl is not None else DEFAULT_TTL,
|
|
}
|
|
write_config(cfg)
|
|
return {"ok": True, "alias": req.alias,
|
|
"note": "In config.yaml geschrieben. llama-swap mit -watch-config laedt automatisch neu."}
|
|
|
|
|
|
@router.post("/update_model")
|
|
def update_model(req: UpdateReq):
|
|
cfg = read_config()
|
|
if req.alias not in cfg.get("models", {}):
|
|
raise HTTPException(404, "Modell nicht gefunden")
|
|
|
|
spec = cfg["models"][req.alias]
|
|
cmd = str(spec.get("cmd", ""))
|
|
|
|
# Replace or add context size
|
|
if re.search(r'-(?:c|-ctx-size)\s+\d+', cmd):
|
|
cmd = re.sub(r'-(?:c|-ctx-size)\s+\d+', f'-c {req.ctx}', cmd)
|
|
else:
|
|
cmd = cmd.strip() + f" -c {req.ctx}\n"
|
|
|
|
cfg["models"][req.alias]["cmd"] = LiteralScalarString(cmd)
|
|
write_config(cfg)
|
|
return {"ok": True}
|
|
|
|
|
|
def _dir_size(p: Path) -> int:
|
|
if p.is_file():
|
|
try:
|
|
return p.stat().st_size
|
|
except OSError:
|
|
return 0
|
|
total = 0
|
|
for root, _dirs, files in os.walk(p):
|
|
for f in files:
|
|
try:
|
|
total += os.path.getsize(os.path.join(root, f))
|
|
except OSError:
|
|
pass
|
|
return total
|
|
|
|
|
|
def _delete_model_files(model_path: str, cfg: dict) -> tuple[str | None, int]:
|
|
"""Loescht die Dateien eines Modells — STRENG auf MODELS_DIR begrenzt. Loescht den
|
|
ganzen Repo-Unterordner (inkl. .cache), wenn dieser ein direktes Kind von MODELS_DIR
|
|
ist und kein anderes (verbleibendes) Modell eine Datei darin nutzt; sonst nur die
|
|
GGUF-Datei selbst. Gibt (geloeschter_pfad|None, freigegebene_bytes) zurueck."""
|
|
root = MODELS_DIR.resolve()
|
|
p = Path(model_path).resolve()
|
|
if root not in p.parents: # Sicherheit: niemals ausserhalb des Modell-Ordners loeschen
|
|
return None, 0
|
|
parent = p.parent
|
|
others = " ".join(str(s.get("cmd", "")) for s in (cfg.get("models") or {}).values())
|
|
parent_shared = parent != root and str(parent) in others
|
|
target = parent if (parent != root and not parent_shared) else p
|
|
if not target.exists():
|
|
return None, 0
|
|
freed = _dir_size(target)
|
|
if target.is_dir():
|
|
shutil.rmtree(target)
|
|
else:
|
|
target.unlink()
|
|
return str(target), freed
|
|
|
|
|
|
@router.post("/delete_model")
|
|
def delete_model(req: DeleteReq):
|
|
"""Modell komplett entfernen: aus der config.yaml austragen (+ entladen) und optional
|
|
die GGUF-Dateien loeschen, um Speicher freizugeben. Unwiderruflich."""
|
|
cfg = read_config()
|
|
models = cfg.get("models", {})
|
|
if req.alias not in models:
|
|
raise HTTPException(404, "Modell nicht gefunden.")
|
|
cmd = str(models[req.alias].get("cmd", ""))
|
|
m = re.search(r'-(?:m|-model)\s+(\S+)', cmd)
|
|
model_path = m.group(1).strip('"\'') if m else None
|
|
|
|
# Erst entladen (falls geladen) — Fehler ignorieren, das Modell soll trotzdem weg.
|
|
try:
|
|
with httpx.Client(timeout=10.0) as c:
|
|
c.post(f"{LLAMA_SWAP_URL}/api/models/unload/{req.alias}")
|
|
except Exception: # noqa: BLE001
|
|
pass
|
|
|
|
del models[req.alias]
|
|
write_config(cfg) # cfg enthaelt das Modell jetzt nicht mehr -> _delete prueft die Restmenge
|
|
|
|
note, freed = "", 0
|
|
if req.delete_files and model_path:
|
|
try:
|
|
deleted, freed = _delete_model_files(model_path, cfg)
|
|
note = (f"Dateien gelöscht ({round(freed / 1024 ** 3, 1)} GB frei)."
|
|
if deleted else "Dateien liegen außerhalb des Modell-Ordners — nur ausgetragen.")
|
|
except Exception as exc: # noqa: BLE001
|
|
note = f"Aus der Konfiguration entfernt, aber Dateien konnten nicht gelöscht werden: {exc}"
|
|
return {"ok": True, "alias": req.alias, "freed_bytes": freed, "note": note or "Modell entfernt."}
|
|
|
|
|
|
@router.post("/unload")
|
|
def unload(model: str | None = None):
|
|
path = f"/api/models/unload/{model}" if model else "/api/models/unload"
|
|
try:
|
|
with httpx.Client(timeout=10.0) as c:
|
|
r = c.post(f"{LLAMA_SWAP_URL}{path}")
|
|
return {"ok": r.status_code < 400, "status": r.status_code}
|
|
except Exception as exc: # noqa: BLE001
|
|
raise HTTPException(502, f"llama-swap nicht erreichbar: {exc}")
|
|
|
|
|
|
@router.post("/chat")
|
|
def chat(req: ChatReq):
|
|
payload = {"model": req.model, "messages": [{"role": "user", "content": req.message}]}
|
|
try:
|
|
with httpx.Client(timeout=120.0) as c:
|
|
r = c.post(f"{LLAMA_SWAP_URL}/v1/chat/completions", json=payload)
|
|
r.raise_for_status()
|
|
data = r.json()
|
|
return {"reply": data["choices"][0]["message"]["content"]}
|
|
except Exception as exc: # noqa: BLE001
|
|
raise HTTPException(502, f"Anfrage fehlgeschlagen: {exc}")
|