Fix: ungültige --prompt-cache-Flags entfernen (blockierten llama-server-Start)
llama-server lehnt --prompt-cache/--prompt-cache-all ab ("invalid argument")
und startet dann nicht — dadurch ließ sich KEIN Modell mehr wecken (u.a. das
Hermes-WebUI bekam "empty stream"). Die Flags gehören zu llama-cli, nicht zum
Server; Prompt-Caching macht llama-server automatisch pro Slot (KV-Reuse).
- config.py: Flags aus _DEFAULT_CMD_TEMPLATE entfernt (+ Warnhinweis).
- migrate_config.py: statt die Flags hinzuzufügen, entfernt es sie nun aus
bestehenden cmds (Reparatur-Migration).
Box-config.yaml wurde bereits direkt korrigiert (alle 7 Modelle); verifiziert:
Hermes lädt + streamt wieder sauber.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
+4
-1
@@ -22,9 +22,12 @@ DISCOVER_TTL = int(os.environ.get("MC_DISCOVER_TTL", "43200")) # 12 h
|
|||||||
# Geteiltes Gedächtnis (SQLite, WAL). Persistent neben den Modellen.
|
# Geteiltes Gedächtnis (SQLite, WAL). Persistent neben den Modellen.
|
||||||
MEMORY_DB = Path(os.environ.get("MC_MEMORY_DB", str(MODELS_DIR / "mc2-memory.db")))
|
MEMORY_DB = Path(os.environ.get("MC_MEMORY_DB", str(MODELS_DIR / "mc2-memory.db")))
|
||||||
# Befehl-Vorlage für llama-swap: {model}=GGUF-Pfad, {ctx}=Kontext, ${PORT} bleibt stehen.
|
# Befehl-Vorlage für llama-swap: {model}=GGUF-Pfad, {ctx}=Kontext, ${PORT} bleibt stehen.
|
||||||
|
# Hinweis: --prompt-cache/--prompt-cache-all sind llama-CLI-Flags, NICHT llama-server —
|
||||||
|
# llama-server lehnt sie ab ("invalid argument") und startet dann nicht. Prompt-Caching
|
||||||
|
# macht llama-server ohnehin automatisch pro Slot (KV-Reuse).
|
||||||
_DEFAULT_CMD_TEMPLATE = (
|
_DEFAULT_CMD_TEMPLATE = (
|
||||||
"llama-server -m {model} --host 127.0.0.1 --port ${PORT} "
|
"llama-server -m {model} --host 127.0.0.1 --port ${PORT} "
|
||||||
"-c {ctx} -ngl 999 -fa 1 --no-mmap --prompt-cache --prompt-cache-all"
|
"-c {ctx} -ngl 999 -fa 1 --no-mmap"
|
||||||
)
|
)
|
||||||
CMD_TEMPLATE = os.environ.get("MC_CMD_TEMPLATE", _DEFAULT_CMD_TEMPLATE)
|
CMD_TEMPLATE = os.environ.get("MC_CMD_TEMPLATE", _DEFAULT_CMD_TEMPLATE)
|
||||||
if "{model}" not in CMD_TEMPLATE:
|
if "{model}" not in CMD_TEMPLATE:
|
||||||
|
|||||||
@@ -27,11 +27,10 @@ def migrate():
|
|||||||
|
|
||||||
print(f"Migrating model: {name}")
|
print(f"Migrating model: {name}")
|
||||||
|
|
||||||
# 1. Ensure prompt caching flags exist
|
# 1. Defektes --prompt-cache/--prompt-cache-all entfernen (llama-CLI-Flags,
|
||||||
if "--prompt-cache " not in cmd and not cmd.endswith("--prompt-cache") and not cmd.endswith("--prompt-cache\n"):
|
# die llama-server ablehnt → Start scheitert). Caching macht llama-server
|
||||||
cmd = cmd.strip() + " --prompt-cache"
|
# automatisch pro Slot.
|
||||||
if "--prompt-cache-all" not in cmd:
|
cmd = cmd.replace(" --prompt-cache-all", "").replace(" --prompt-cache", "")
|
||||||
cmd = cmd.strip() + " --prompt-cache-all"
|
|
||||||
|
|
||||||
# 2. Extract aliases/role
|
# 2. Extract aliases/role
|
||||||
aliases = spec.get("aliases", [])
|
aliases = spec.get("aliases", [])
|
||||||
|
|||||||
Reference in New Issue
Block a user