""" Engine-Service: liest die llama-swap config.yaml (read-only in Phase 0) und spricht die llama-swap-API (/v1/models, /running). Schreiblogik (Modelle installieren, Gruppen/Routing verwalten) kommt in Phase 1. Logik portiert aus Mission Control v1 (llamaswap.py + routers/models.py), auf das Nötigste reduziert. """ import os import re import httpx from config import CONFIG_PATH, LLAMA_SWAP_URL, yaml # Kanonische Rollen (vereinheitlicht ggü. v1: kein manager/reviewer mehr). # Eine Quelle der Wahrheit — Capability-Erkennung läuft separat über model_caps (Phase 1). ROLE_IDS = {"vision", "coder", "reasoning", "agent", "scout"} _CTX_RE = re.compile(r"-(?:c|-ctx-size)\s+(\d+)") _PATH_RE = re.compile(r"-(?:m|-model)\s+([^\s]+)") _QUANT_RE = re.compile(r"(Q\d_[A-Z0-9_]+|IQ\d_[A-Z0-9_]+|fp16|bf16)\.gguf", re.IGNORECASE) def read_config() -> dict: """llama-swap config.yaml laden. Existiert sie nicht (z.B. lokaler Dev-PC ohne Engine), wird ein leeres Modell-Set zurückgegeben statt zu werfen.""" if not CONFIG_PATH.exists(): return {"models": {}} with CONFIG_PATH.open("r", encoding="utf-8") as f: data = yaml.load(f) or {} if not data.get("models"): data["models"] = {} return data def _parse_model(name: str, spec: dict) -> dict: """Ein config.yaml-Modell in ein flaches UI-Objekt übersetzen.""" spec = spec or {} cmd = str(spec.get("cmd", "")).strip() ctx = None if (m := _CTX_RE.search(cmd)): ctx = int(m.group(1)) path = "" filename = "" quant = "" size_bytes = None if (m := _PATH_RE.search(cmd)): path = m.group(1).replace("'", "").replace('"', "") filename = os.path.basename(path) if os.path.exists(path): size_bytes = os.path.getsize(path) if (q := _QUANT_RE.search(path)): quant = q.group(1).upper() aliases = spec.get("aliases") or [] if isinstance(aliases, str): aliases = [aliases] aliases = [str(a) for a in aliases] # Rolle = erster Alias; Legacy-Fallback: Key selbst ist eine Rolle. role = aliases[0].lower() if aliases else (name.lower() if name.lower() in ROLE_IDS else None) return { "name": name, "role": role, "aliases": aliases, "api_ids": [name] + aliases, "ctx": ctx, "ttl": spec.get("ttl"), "cmd": cmd, "filename": filename, "quant": quant, "size_bytes": size_bytes, # "incomplete" = Eintrag ohne hinterlegtes Modell (-m), z.B. Platzhalter. "incomplete": not path, } def list_models() -> list[dict]: """Alle in der config.yaml konfigurierten Modelle (read-only).""" cfg = read_config() return [_parse_model(name, spec) for name, spec in (cfg.get("models") or {}).items()] def engine_reachable() -> bool: """Ist die llama-swap-API erreichbar?""" try: with httpx.Client(timeout=3.0) as c: return c.get(f"{LLAMA_SWAP_URL}/v1/models").status_code == 200 except Exception: return False