""" Routing-Gateway-Status (eingebauter Modus). MC2 IST der Gateway: serviert `/v1/*` mit `model: auto`-Komplexitäts-Routing vor llama-swap. Kein externer LiteLLM-Dienst nötig (baut auf Python 3.14 nicht); bleibt später austauschbar. """ from config import PORT from services.llamaswap import engine_reachable from services.routing_policy import load_policy def routing_summary() -> dict: p = load_policy() coding_default = p["coder_lite"] or p["coder"] return { "mode": "builtin", "endpoint": f":{PORT}/v1 (OpenAI-kompatibel)", # Virtuelle Lanes, die Clients/IDEs als „Modell" wählen (Router pickt das echte Alias). "lanes": [ { "name": "chat", "aka": "auto", "target": f"{p['fast']} ↔ {p['heavy']} (nach Komplexität)", "threshold_chars": p["heavy_chars"], }, { "name": "coding", "target": f"{coding_default} ↔ {p['coder']} (Eskalation)", "escalate_chars": p["coding_escalate_chars"], }, ], # Rückwärtskompatible Flach-Liste (alte UI/Clients). "routes": [ {"name": "chat", "target": f"{p['fast']} ↔ {p['heavy']} (nach Komplexität)"}, {"name": "coding", "target": f"{coding_default} ↔ {p['coder']} (Eskalation)"}, {"name": "", "target": "llama-swap-Passthrough (lädt bei Bedarf)"}, ], "heavy_threshold_chars": p["heavy_chars"], "fallbacks": [], "context_window_fallbacks": [], } def gateway_reachable() -> bool: # Der eingebaute Gateway lebt in MC und proxyt llama-swap → erreichbar, wenn Engine läuft. return engine_reachable()