feat(2.0): W1 — fast-Spur ohne Thinking (flotte Antworten) + tool-call-Cap
Gateway injiziert chat_template_kwargs.enable_thinking=false fuer die fast- Spur (Qwen3.6 ist Reasoning-Modell → sonst lahm/leer). heavy behaelt Thinking. Env MC_FAST_NO_THINK. Hermes-Thrash war poisoned Persistent- Session (fresh=clean, 34k statt 249k verifiziert); code_execution.max_tool_ calls 50->20 auf der Box (Historie unangetastet). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
+1
-1
@@ -52,7 +52,7 @@ PORT = int(os.environ.get("MC_PORT", "9000"))
|
|||||||
FRONTEND_DIST = Path(os.environ.get("MC_FRONTEND_DIST", str(Path(__file__).resolve().parent.parent / "frontend" / "dist")))
|
FRONTEND_DIST = Path(os.environ.get("MC_FRONTEND_DIST", str(Path(__file__).resolve().parent.parent / "frontend" / "dist")))
|
||||||
|
|
||||||
# Version (Phase 0 — Greenfield-Skeleton).
|
# Version (Phase 0 — Greenfield-Skeleton).
|
||||||
VERSION = "2.0.0-phase6"
|
VERSION = "2.0.0-w1"
|
||||||
|
|
||||||
# Gemeinsame YAML-Instanz (preserve_quotes hält Kommentare/Quotes in config.yaml).
|
# Gemeinsame YAML-Instanz (preserve_quotes hält Kommentare/Quotes in config.yaml).
|
||||||
yaml = YAML()
|
yaml = YAML()
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ from fastapi import APIRouter, Request
|
|||||||
from fastapi.responses import JSONResponse, StreamingResponse
|
from fastapi.responses import JSONResponse, StreamingResponse
|
||||||
|
|
||||||
from config import LLAMA_SWAP_URL
|
from config import LLAMA_SWAP_URL
|
||||||
from services.router_logic import choose_model
|
from services.router_logic import FAST, FAST_NO_THINK, choose_model
|
||||||
|
|
||||||
router = APIRouter(prefix="/v1")
|
router = APIRouter(prefix="/v1")
|
||||||
|
|
||||||
@@ -32,7 +32,11 @@ async def _proxy(path: str, request: Request):
|
|||||||
body["model"] = alias
|
body["model"] = alias
|
||||||
routed = {"x-mc-routed-to": alias, "x-mc-route-reason": reason}
|
routed = {"x-mc-routed-to": alias, "x-mc-route-reason": reason}
|
||||||
else:
|
else:
|
||||||
|
alias = requested
|
||||||
routed = {"x-mc-routed-to": requested}
|
routed = {"x-mc-routed-to": requested}
|
||||||
|
# fast-Spur: Thinking aus für flotte Antworten (sofern Client es nicht selbst setzt).
|
||||||
|
if FAST_NO_THINK and alias == FAST and "chat_template_kwargs" not in body:
|
||||||
|
body["chat_template_kwargs"] = {"enable_thinking": False}
|
||||||
url = f"{LLAMA_SWAP_URL}{path}"
|
url = f"{LLAMA_SWAP_URL}{path}"
|
||||||
|
|
||||||
if body.get("stream"):
|
if body.get("stream"):
|
||||||
|
|||||||
@@ -9,6 +9,9 @@ import re
|
|||||||
FAST = os.environ.get("MC_ROUTE_FAST", "fast")
|
FAST = os.environ.get("MC_ROUTE_FAST", "fast")
|
||||||
HEAVY = os.environ.get("MC_ROUTE_HEAVY", "heavy")
|
HEAVY = os.environ.get("MC_ROUTE_HEAVY", "heavy")
|
||||||
HEAVY_CHARS = int(os.environ.get("MC_GATEWAY_HEAVY_CHARS", "8000"))
|
HEAVY_CHARS = int(os.environ.get("MC_GATEWAY_HEAVY_CHARS", "8000"))
|
||||||
|
# Thinking auf der fast-Spur ausschalten → flotte Alltags-Antworten (Qwen3.6 ist ein
|
||||||
|
# Reasoning-Modell). heavy behält Thinking für harte Aufgaben. Abschaltbar via Env.
|
||||||
|
FAST_NO_THINK = os.environ.get("MC_FAST_NO_THINK", "1") not in ("0", "false", "")
|
||||||
|
|
||||||
_HEAVY_KW = re.compile(
|
_HEAVY_KW = re.compile(
|
||||||
r"\b(beweis|prove|theorem|refactor|architect|komplex|complex|schwierig|"
|
r"\b(beweis|prove|theorem|refactor|architect|komplex|complex|schwierig|"
|
||||||
|
|||||||
Reference in New Issue
Block a user