501ba36b89
Gateway injiziert chat_template_kwargs.enable_thinking=false fuer die fast- Spur (Qwen3.6 ist Reasoning-Modell → sonst lahm/leer). heavy behaelt Thinking. Env MC_FAST_NO_THINK. Hermes-Thrash war poisoned Persistent- Session (fresh=clean, 34k statt 249k verifiziert); code_execution.max_tool_ calls 50->20 auf der Box (Historie unangetastet). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
63 lines
2.2 KiB
Python
63 lines
2.2 KiB
Python
"""
|
|
Eingebauter Routing-Gateway (OpenAI-kompatibel) — EIN Endpunkt für Hermes + IDEs.
|
|
|
|
`model: auto` → Komplexitäts-Routing fast↔heavy; jeder andere Name geht als
|
|
llama-swap-Alias durch (das lädt das Modell bei Bedarf). Streaming wird
|
|
durchgereicht. Ersetzt LiteLLM (das auf Python 3.14 nicht baut) — gleicher
|
|
Vertrag, später austauschbar.
|
|
"""
|
|
|
|
import httpx
|
|
from fastapi import APIRouter, Request
|
|
from fastapi.responses import JSONResponse, StreamingResponse
|
|
|
|
from config import LLAMA_SWAP_URL
|
|
from services.router_logic import FAST, FAST_NO_THINK, choose_model
|
|
|
|
router = APIRouter(prefix="/v1")
|
|
|
|
|
|
@router.get("/models")
|
|
async def models():
|
|
async with httpx.AsyncClient(timeout=10) as c:
|
|
r = await c.get(f"{LLAMA_SWAP_URL}/v1/models")
|
|
return JSONResponse(r.json(), status_code=r.status_code)
|
|
|
|
|
|
async def _proxy(path: str, request: Request):
|
|
body = await request.json()
|
|
requested = str(body.get("model") or "auto")
|
|
if requested == "auto":
|
|
alias, reason = choose_model(body)
|
|
body["model"] = alias
|
|
routed = {"x-mc-routed-to": alias, "x-mc-route-reason": reason}
|
|
else:
|
|
alias = requested
|
|
routed = {"x-mc-routed-to": requested}
|
|
# fast-Spur: Thinking aus für flotte Antworten (sofern Client es nicht selbst setzt).
|
|
if FAST_NO_THINK and alias == FAST and "chat_template_kwargs" not in body:
|
|
body["chat_template_kwargs"] = {"enable_thinking": False}
|
|
url = f"{LLAMA_SWAP_URL}{path}"
|
|
|
|
if body.get("stream"):
|
|
async def gen():
|
|
async with httpx.AsyncClient(timeout=None) as c:
|
|
async with c.stream("POST", url, json=body) as r:
|
|
async for chunk in r.aiter_raw():
|
|
yield chunk
|
|
return StreamingResponse(gen(), media_type="text/event-stream", headers=routed)
|
|
|
|
async with httpx.AsyncClient(timeout=600) as c:
|
|
r = await c.post(url, json=body)
|
|
return JSONResponse(r.json(), status_code=r.status_code, headers=routed)
|
|
|
|
|
|
@router.post("/chat/completions")
|
|
async def chat_completions(request: Request):
|
|
return await _proxy("/v1/chat/completions", request)
|
|
|
|
|
|
@router.post("/completions")
|
|
async def completions(request: Request):
|
|
return await _proxy("/v1/completions", request)
|