diff --git a/llamaswap.py b/llamaswap.py index 264299c..9fb896a 100644 --- a/llamaswap.py +++ b/llamaswap.py @@ -7,11 +7,52 @@ Helfer rund um llama-swap und dessen config.yaml. """ import os +import re import httpx from config import CONFIG_PATH, LLAMA_SWAP_URL, yaml +# Kanonische Rollen (Tags auf den Modellen). Bewusst eine kleine, feste Palette fuer die +# UI-Schnellwahl — frei waehlbare Rollen sind trotzdem erlaubt. +ROLE_IDS = {"vision", "coder", "scout", "reviewer", "manager"} + + +def model_id_from_path(model_path: str) -> str: + """Sprechende Modell-ID (= llama-swap-Modellname, den man in der API angibt) aus dem + GGUF-Pfad ableiten: der Repo-Ordnername ohne '-GGUF'. Fallback: Dateiname ohne + Quant-Suffix/.gguf. So steht in der Modell-Liste der echte Name statt 'coder'.""" + d = os.path.basename(os.path.dirname(model_path)) + name = re.sub(r"[-_]?GGUF$", "", d, flags=re.I).strip("-_") + if not name: + fn = re.sub(r"\.gguf$", "", os.path.basename(model_path), flags=re.I) + fn = re.sub(r"-\d+-of-\d+$", "", fn) + name = re.sub(r"[-_](Q\d[\w]*|IQ\d[\w]*|F16|BF16|FP16|F32)$", "", fn, flags=re.I) + return name or "modell" + + +def set_role_alias(cfg: dict, model_id: str, role: str | None) -> None: + """Rolle als llama-swap-`aliases` auf ein Modell legen (damit der Rollenname in der API + ebenfalls funktioniert). Haelt den Alias EINDEUTIG: derselbe Rollen-Alias wird vorher + bei allen anderen Modellen entfernt. role=None/leer entfernt den Alias.""" + models = cfg.get("models") or {} + role = (role or "").strip().lower() + if role: + for mid, spec in models.items(): + if mid == model_id or not isinstance(spec, dict): + continue + al = [a for a in (spec.get("aliases") or []) if str(a).lower() != role] + if al: + spec["aliases"] = al + else: + spec.pop("aliases", None) + spec = models.get(model_id) + if isinstance(spec, dict): + if role and role != model_id.lower(): + spec["aliases"] = [role] + else: + spec.pop("aliases", None) + def _swap_get(path: str): with httpx.Client(timeout=5.0) as c: diff --git a/routers/cookbook.py b/routers/cookbook.py index 6ebb72f..219c3e8 100644 --- a/routers/cookbook.py +++ b/routers/cookbook.py @@ -17,7 +17,7 @@ from auth import auth from hw_math import evaluate_fit, max_ctx_for from config import (MODELS_DIR, CMD_TEMPLATE, DEFAULT_TTL, HF_DOWNLOAD_ENV, USER_RECIPES_PATH, DISCOVER_CACHE_PATH, DISCOVER_TTL, hf_bin) -from llamaswap import read_config, write_config +from llamaswap import read_config, write_config, model_id_from_path, set_role_alias from jobengine import start_job, JOBS, attach_download_progress from recipes import RECIPES, UPGRADES from sources import TRUSTED_AUTHORS, CATEGORIES, SKIP_TOKENS @@ -255,7 +255,10 @@ def install_recipe(req: InstallRecipeReq): ctx = min(max_ctx_for(m["params_b"], m["quant"], ram_gb), 32768) path = str(target / file) cmd = CMD_TEMPLATE.replace("{model}", path).replace("{ctx}", str(ctx)) - cfg["models"][m["role"]] = {"cmd": LiteralScalarString(cmd + "\n"), "ttl": DEFAULT_TTL} + # Schluessel = sprechender Modellname; Rolle (aus dem Rezept) als Alias. + mid = model_id_from_path(path) + cfg["models"][mid] = {"cmd": LiteralScalarString(cmd + "\n"), "ttl": DEFAULT_TTL} + set_role_alias(cfg, mid, m["role"]) write_config(cfg) return {"job_ids": job_ids, "count": len(job_ids)} @@ -484,8 +487,11 @@ def install_model(req: InstallModelReq): attach_download_progress(jid, str(target), hf_file_size(req.repo, file)) cfg = read_config() ctx = min(max_ctx_for(req.params_b, req.quant, ram_gb), 32768) - cmd = CMD_TEMPLATE.replace("{model}", str(target / file)).replace("{ctx}", str(ctx)) - cfg["models"][req.role] = {"cmd": LiteralScalarString(cmd + "\n"), "ttl": DEFAULT_TTL} + path = str(target / file) + cmd = CMD_TEMPLATE.replace("{model}", path).replace("{ctx}", str(ctx)) + mid = model_id_from_path(path) + cfg["models"][mid] = {"cmd": LiteralScalarString(cmd + "\n"), "ttl": DEFAULT_TTL} + set_role_alias(cfg, mid, req.role) write_config(cfg) - return {"job_id": jid} + return {"job_id": jid, "model_id": mid} diff --git a/routers/models.py b/routers/models.py index 4259af4..068b61c 100644 --- a/routers/models.py +++ b/routers/models.py @@ -17,7 +17,8 @@ from config import (CMD_TEMPLATE, CONFIG_PATH, DEFAULT_TTL, HF_DOWNLOAD_ENV, LLA MODELS_DIR, TOKEN, hf_bin) from jobengine import JOBS, start_job, attach_download_progress from routers.cookbook import hf_file_size -from llamaswap import _swap_get, read_config, write_config +from llamaswap import (_swap_get, read_config, write_config, + model_id_from_path, set_role_alias, ROLE_IDS) from hw_math import extract_params_b, max_ctx_for, estimate_memory_gb import re import os @@ -38,12 +39,18 @@ class DownloadReq(BaseModel): class RegisterReq(BaseModel): - alias: str + alias: str = "" # rueckwaertskompatibel: wird als Rolle interpretiert, wenn 'role' fehlt + role: str | None = None # Rollen-Tag (vision/coder/scout/reviewer/manager o.ae.) model_path: str ctx: int = 8192 ttl: int | None = None +class RoleReq(BaseModel): + alias: str # die Modell-ID (config-Key) + role: str = "" # leer = Rolle entfernen + + class ChatReq(BaseModel): model: str message: str @@ -92,17 +99,28 @@ def status(): if q_match: quant = q_match.group(1).upper() + # Rolle aus dem llama-swap-Alias ableiten; Legacy-Eintraege ohne Alias, deren Key + # selbst eine Rolle ist (coder/vision/...), behalten diese als Rolle. + aliases = spec.get("aliases") or [] + if isinstance(aliases, str): + aliases = [aliases] + aliases = [str(a) for a in aliases] + role = aliases[0].lower() if aliases else (name.lower() if name.lower() in ROLE_IDS else None) + caps = ["Text"] - if "coder" in name.lower() or (m_path and "code" in m_path.group(1).lower()): + if role == "coder" or "coder" in name.lower() or (m_path and "code" in m_path.group(1).lower()): caps = ["Code"] if "--mmproj" in cmd: caps.append("Bild") - + # "incomplete" = Rolle existiert in der config, hat aber kein Modell (-m) hinterlegt # (z.B. von provision.sh angelegter Platzhalter). Ehrlich kennzeichnen statt "bereit". incomplete = not (m_path and m_path.group(1)) configured[name] = { "name": name, + "role": role, + "aliases": aliases, + "api_ids": [name] + aliases, "ttl": spec.get("ttl", cfg.get("globalTTL", 0)), "cmd": cmd, "state": "idle", @@ -184,15 +202,31 @@ def register(req: RegisterReq): cfg = read_config() cmd = CMD_TEMPLATE.replace("{model}", req.model_path).replace("{ctx}", str(req.ctx)) cmd = _augment_vision(cmd, req.model_path) - cfg["models"][req.alias] = { + # Neues Schema: Schluessel = sprechender Modellname (steht so in der Modell-Liste und ist + # der API-Name), die Rolle kommt als llama-swap-Alias obendrauf (beide Namen funktionieren). + role = (req.role or req.alias or "").strip().lower() + model_id = model_id_from_path(req.model_path) + cfg["models"][model_id] = { "cmd": LiteralScalarString(cmd + "\n"), "ttl": req.ttl if req.ttl is not None else DEFAULT_TTL, } + set_role_alias(cfg, model_id, role) write_config(cfg) - return {"ok": True, "alias": req.alias, + return {"ok": True, "alias": model_id, "model_id": model_id, "role": role, "note": "In config.yaml geschrieben. llama-swap mit -watch-config laedt automatisch neu."} +@router.post("/set_role") +def set_role(req: RoleReq): + """Rollen-Tag eines Modells setzen/aendern (als eindeutiger llama-swap-Alias).""" + cfg = read_config() + if req.alias not in cfg.get("models", {}): + raise HTTPException(404, "Modell nicht gefunden.") + set_role_alias(cfg, req.alias, req.role.strip().lower() or None) + write_config(cfg) + return {"ok": True, "alias": req.alias, "role": req.role.strip().lower()} + + @router.post("/update_model") def update_model(req: UpdateReq): cfg = read_config() diff --git a/static/js/panels/cookbook.js b/static/js/panels/cookbook.js index 97fad19..ee5bc16 100644 --- a/static/js/panels/cookbook.js +++ b/static/js/panels/cookbook.js @@ -109,7 +109,7 @@ function mount() {
${esc(id)}`).join('·')}
+