Compare commits
214 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 7704ffddab | |||
| 588d032335 | |||
| e84386b03e | |||
| a16287e19c | |||
| f88c2a95c1 | |||
| ed55612cfa | |||
| 47ade040e6 | |||
| 852e723cc8 | |||
| f90e5f4519 | |||
| c3d1cc87e2 | |||
| 88c2659a9c | |||
| 6f7498a956 | |||
| a1cea1a1ea | |||
| 60765469aa | |||
| 68ad291a8f | |||
| e08d812598 | |||
| 45fb61000a | |||
| 403961da51 | |||
| 64efe18500 | |||
| 98360ad31f | |||
| 09a1c98514 | |||
| aff0105700 | |||
| 2db839aef9 | |||
| 9967155332 | |||
| 3c0a24feb7 | |||
| 38f8c96b66 | |||
| c3c7c2ca91 | |||
| feba73a11a | |||
| 16f4683859 | |||
| 41f46569bf | |||
| ac475ba509 | |||
| ef1e7eb8e6 | |||
| cb4d7102b5 | |||
| 984224959a | |||
| f655f09ce1 | |||
| dd99401f3e | |||
| 087eb2259d | |||
| 8108d3fd28 | |||
| a894a4e952 | |||
| c1fac1e688 | |||
| 9e432dbd2d | |||
| 3611defe5d | |||
| c77be502f9 | |||
| ec9d1bff5b | |||
| bb922b2a21 | |||
| 8a802241f7 | |||
| a9a26cf856 | |||
| b31736cde7 | |||
| 50e04e0aee | |||
| 68cad9c0f1 | |||
| 9c3dd9be82 | |||
| d1ea505b7a | |||
| 63a83c69c3 | |||
| 181db38b49 | |||
| b1fa8bea43 | |||
| d73075bea0 | |||
| 71d1d99c44 | |||
| 4ee016d3a7 | |||
| 9923743219 | |||
| 637cbd9135 | |||
| 88661545b4 | |||
| de3c452f20 | |||
| d832f90ad6 | |||
| 7bc20a6302 | |||
| e2b3bb7088 | |||
| 209afebefa | |||
| 9332971384 | |||
| 799a0c72e9 | |||
| d07fe1de66 | |||
| 49c9e27502 | |||
| b4a8f92cfa | |||
| 15d5598eec | |||
| dd4c4c8c0f | |||
| 6d7a17e765 | |||
| 53d0796bd5 | |||
| eeb397f066 | |||
| 75727c6b36 | |||
| f08595910d | |||
| 11f0066b47 | |||
| aa62c98247 | |||
| cf587616dd | |||
| 8e7ce1b1d3 | |||
| 2360ad173a | |||
| f82729f88e | |||
| d3157d2535 | |||
| 58dda66f84 | |||
| b38e3360c5 | |||
| 46f108b6f3 | |||
| 8554e7b29c | |||
| 9fb321e45e | |||
| b383711f6d | |||
| 689ea48d72 | |||
| 00209dedff | |||
| 56243e1835 | |||
| 5c3f50dfa5 | |||
| 965d7b2002 | |||
| 6d552b035a | |||
| 05bef8642d | |||
| 019f08093d | |||
| 45d635afae | |||
| 03933ade35 | |||
| dcfede7e69 | |||
| cf65589b93 | |||
| 707b812f8b | |||
| 694aff9801 | |||
| 89cdc60b6e | |||
| ec9488d94e | |||
| 35189fde0e | |||
| ea256fca0d | |||
| 3dc878f012 | |||
| 578d09ac7c | |||
| c4708d7d5d | |||
| 763e634dfd | |||
| 1e011714dd | |||
| 6d529e3f79 | |||
| bfa6844124 | |||
| 14325e7690 | |||
| c074d977ce | |||
| 123303240d | |||
| 31d5e5d727 | |||
| 6758bbfbd9 | |||
| 45e74a97bd | |||
| 46777cb99a | |||
| 75a1be4a71 | |||
| 43880b1965 | |||
| 5b699aaa79 | |||
| 510de69250 | |||
| 8bb4e11f31 | |||
| 38f0394166 | |||
| 65d8ab5fe3 | |||
| c48e583790 | |||
| 530d77ff1b | |||
| 463aae19a9 | |||
| 45bede6579 | |||
| 0afddcdfdc | |||
| 6a464bf653 | |||
| 483eb0b2fb | |||
| a64b30c429 | |||
| 8881d65c8d | |||
| 037c4b11df | |||
| b51fc899ca | |||
| e19831f7d5 | |||
| 9b68db9e82 | |||
| 879afbb1d4 | |||
| abd9392c90 | |||
| 689bf3eed3 | |||
| b51f30b49a | |||
| 1e8e114550 | |||
| 93111c6eaf | |||
| e2547ec301 | |||
| 825fd60972 | |||
| eafeaf333d | |||
| f1b0d61ada | |||
| 7a11fd2846 | |||
| 0266dc9e92 | |||
| 74f64731ab | |||
| 341ea870bb | |||
| a7c3f8f516 | |||
| 35dcc69ba5 | |||
| 2c60caf790 | |||
| 2e4cddc840 | |||
| 311f4d7b68 | |||
| 77daa38cf0 | |||
| d286dbf203 | |||
| 066feee3ea | |||
| 0c7b0b19af | |||
| 563e7837b9 | |||
| f32e4baf5b | |||
| da929845a9 | |||
| 4acea97f03 | |||
| 18f361d485 | |||
| dbe6e4b4f3 | |||
| 42d2e58570 | |||
| 1e879ca3c4 | |||
| 3fa23d16c6 | |||
| bc3abb127a | |||
| b90cef3968 | |||
| e88dfb8f98 | |||
| b6af1c1b4a | |||
| e9fb72fdc7 | |||
| 8d5885d683 | |||
| f1e0503c73 | |||
| 97f8bc34bc | |||
| be5341fb03 | |||
| 85d8371261 | |||
| ef93b7f919 | |||
| b08e867c1a | |||
| bd6aacf3ed | |||
| cb02ede5fb | |||
| d15744a812 | |||
| 6a8e55cc43 | |||
| e1da5c797d | |||
| 807c2c6194 | |||
| 2536d91430 | |||
| 1f391644ca | |||
| 2ea3d01b58 | |||
| c863f01a78 | |||
| 501ba36b89 | |||
| 77b6dee02f | |||
| ba435fb1d7 | |||
| ceca2ae8e3 | |||
| db1f62227b | |||
| fc0153d0de | |||
| af46a7b041 | |||
| f1cbfa8e67 | |||
| 2e6655c398 | |||
| 1b332f86e6 | |||
| 1f4c987652 | |||
| c773dd7eae | |||
| 81468df9c0 | |||
| 6c8b6d81fe | |||
| cff3f0b1a8 | |||
| 1b421e30f9 | |||
| b805c294eb |
@@ -0,0 +1,28 @@
|
||||
{
|
||||
"version": "0.0.1",
|
||||
"configurations": [
|
||||
{
|
||||
"name": "mc2",
|
||||
"runtimeExecutable": "F:\\Coding Stuff\\mission-control-2\\backend\\.venv\\Scripts\\python.exe",
|
||||
"runtimeArgs": [
|
||||
"-m",
|
||||
"uvicorn",
|
||||
"app:app",
|
||||
"--app-dir",
|
||||
"F:\\Coding Stuff\\mission-control-2\\backend",
|
||||
"--port",
|
||||
"9000"
|
||||
],
|
||||
"port": 9000
|
||||
},
|
||||
{
|
||||
"name": "frontend",
|
||||
"runtimeExecutable": "npm",
|
||||
"runtimeArgs": ["run", "dev", "--", "--port", "5180", "--strictPort"],
|
||||
"cwd": "F:\\Coding Stuff\\mission-control-2\\frontend",
|
||||
"env": { "MC_API_TARGET": "http://192.168.178.151:9001" },
|
||||
"autoPort": false,
|
||||
"port": 5180
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
# Shell-/Deploy-Skripte MÜSSEN LF behalten — sonst bricht der Deploy auf der Box
|
||||
# (CRLF macht `set -euo pipefail` zu `pipefail\r` → "invalid option name").
|
||||
*.sh text eol=lf
|
||||
*.bash text eol=lf
|
||||
|
||||
# systemd-Units und Service-Configs ebenfalls LF.
|
||||
*.service text eol=lf
|
||||
*.timer text eol=lf
|
||||
|
||||
# Windows-Batch-Wrapper bleiben CRLF.
|
||||
*.cmd text eol=crlf
|
||||
*.bat text eol=crlf
|
||||
+44
@@ -0,0 +1,44 @@
|
||||
# Python
|
||||
backend/.venv/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
# Node / Vite
|
||||
frontend/node_modules/
|
||||
# frontend/dist wird committet (kein Node-Build auf der Box) — siehe deploy/
|
||||
|
||||
# Avatar-VRM (groß + lizenz-/redistributionssensibel) — liegt lokal + auf der Box, nicht in git.
|
||||
# Wird per Direkt-Deploy auf die Box gespielt (dist/avatar.vrm), nicht über git.
|
||||
frontend/public/avatar.vrm
|
||||
frontend/dist/avatar.vrm
|
||||
|
||||
# Env / local
|
||||
*.env
|
||||
.DS_Store
|
||||
|
||||
# Box-Recon-/Scratch-Skripte (lokale Diagnose, nicht fürs Repo)
|
||||
box_recon*
|
||||
gemma_swap*
|
||||
|
||||
# Lucy-TTS/F5-Experimente: nur Code/Batch committen — venvs, Modelle, Audio, Logs bleiben lokal
|
||||
client/lucy-tts/ptts-venv/
|
||||
client/lucy-tts/venv/
|
||||
client/lucy-tts/llamacpp-vulkan/
|
||||
client/lucy-tts/llamacpp-vulkan.zip
|
||||
client/lucy-tts/models/
|
||||
client/lucy-tts/out_*/
|
||||
client/lucy-tts/*.wav
|
||||
client/lucy-tts/*.mp3
|
||||
client/lucy-tts/*.log
|
||||
client/lucy-tts/*.safetensors
|
||||
client/lucy-tts/*.json
|
||||
client/lucy-f5/.cache/
|
||||
client/lucy-f5/F5-TTS-ONNX/
|
||||
client/lucy-f5/onnx_de/
|
||||
client/lucy-f5/onnx_f32/
|
||||
client/lucy-f5/vocos-mel-24khz/
|
||||
client/lucy-f5/*.onnx
|
||||
client/lucy-f5/*.safetensors
|
||||
client/lucy-f5/*.wav
|
||||
client/lucy-f5/*.zip
|
||||
client/lucy-f5/venv/
|
||||
@@ -0,0 +1,68 @@
|
||||
# Mission Control 2.0
|
||||
|
||||
Komponierbarer Local-AI-Stack für den Bosgame M5. Greenfield-Neuaufbau —
|
||||
siehe Architektur-Plan (`docs/` bzw. der genehmigte Plan).
|
||||
|
||||
**Schichten:** Engine (llama-swap, **Vulkan/RADV** auf Strix Halo) · **Builtin-Routing-Gateway**
|
||||
in MC2 (`model: auto`, kein externer LiteLLM-Dienst — scheitert auf Python 3.14) ·
|
||||
Mission Control 2.0 (FastAPI + React/shadcn) · Hermes Agent + hermes-webui ·
|
||||
Shared Memory (SQLite via MCP). Jede Schicht hinter stabilem Vertrag austauschbar.
|
||||
|
||||
## Status: Phasen 0–5 ✅ · MC2 **live auf der Box** (:9001) · Modelle/Hermes-Wiring + Cutover offen
|
||||
|
||||
Fortschritt & Resume-Guide: siehe [`docs/STATUS.md`](docs/STATUS.md).
|
||||
|
||||
- **Phase 0** — FastAPI-Skeleton + React/shadcn-Shell (Cmd+K, Dark, PWA).
|
||||
- **Phase 1** — Compute-Module (fit/caps/sources, portiert), **Discover** (live HF + Fit + Caps),
|
||||
**Engine-Write** (register + `groups`/Ko-Residenz + vocab-geprüfte Spec-Drafts),
|
||||
**Builtin-Gateway** (`model: auto` + Fallbacks), Frontend **Modelle & Routing** (Caps-Chips, Fit, Discover, Routing-View).
|
||||
|
||||
- **Phase 2** — System-Status (CPU/RAM/GPU/Disk), Wartung (restart/self-update, sudo-frei),
|
||||
**Connect** (saubere IDE-Snippets → Gateway `model:auto`, LAN-IP-Override).
|
||||
- **Phase 3** — Geteiltes **Gedächtnis** (SQLite/WAL, 5 Kategorien, Dedupe-Kurator) + **MCP-Server**
|
||||
(`mcp/mcp_memory.py` shared, `mcp/mcp_mc.py` Stack-Management für Hermes), MemoryView.
|
||||
|
||||
- **Phase 4** — Hermes-**Agent-Status** (`/api/agent/status`, AgentView mit Tiles + „Hermes öffnen"),
|
||||
`deploy/hermes-webui.service`, **Box-Runbook** [`docs/HERMES_SETUP.md`](docs/HERMES_SETUP.md)
|
||||
(hermes-webui, Brain=`auto`, Tools/MCP-Verdrahtung). Box-Ausführung steht noch aus.
|
||||
|
||||
- **Phase 5** — **Backup** (Memory + Configs), **Services-Health** + Observability-Links,
|
||||
**Theme-Toggle** (Hell/Dunkel). Box-Deploy/-Wiring + Cutover (Phase 6) brauchen die Box.
|
||||
|
||||
API: `health · models · discover · fit · models/register · groups · routing · system/* · connect ·
|
||||
memory/* · agent/status` (Details in `docs/STATUS.md`). MCP: `mcp/` (siehe `mcp/requirements.txt`).
|
||||
Box-Runbooks: `docs/HERMES_SETUP.md` + `deploy/` (Units, deploy.sh, backup.sh).
|
||||
|
||||
## Entwickeln
|
||||
|
||||
**Backend:**
|
||||
```bash
|
||||
cd backend
|
||||
python -m venv .venv && .venv/Scripts/python -m pip install -r requirements.txt # Windows
|
||||
.venv/Scripts/python -m uvicorn app:app --port 9000
|
||||
```
|
||||
|
||||
**Frontend (Dev, proxyt /api → :9000):**
|
||||
```bash
|
||||
cd frontend
|
||||
npm install
|
||||
npm run dev # http://localhost:5173
|
||||
```
|
||||
|
||||
**Frontend (Build → wird vom Backend ausgeliefert):**
|
||||
```bash
|
||||
cd frontend && npm run build # → frontend/dist
|
||||
```
|
||||
|
||||
## Env-Vars (Auswahl)
|
||||
|
||||
| Variable | Default | Zweck |
|
||||
|---|---|---|
|
||||
| `MC_LLAMA_SWAP_URL` | `http://127.0.0.1:8080` | Engine |
|
||||
| `MC_CONFIG_PATH` | `/etc/llama-swap/config.yaml` | llama-swap Config |
|
||||
| `MC_GATEWAY_URL` | `http://127.0.0.1:$MC_PORT` | Builtin-Gateway (Teil von MC2, kein externer Dienst) |
|
||||
| `MC_PORT` | `9000` | MC-Backend-Port |
|
||||
| `MC_ENGINE_PATH` | `/opt/llamacpp-vulkan` | Aktive Engine-Binary (Vulkan-Build) |
|
||||
| `MC_ENGINE_REPO` | `ggml-org/llama.cpp` | Quelle für Engine-Update-Check |
|
||||
| `MC_DRAFTS_DIR` | `$MODELS/drafts` | Spec-Draft-Modelle (vocab-geprüft) |
|
||||
| `MC_SPEC_TYPE` | `draft-simple` | Speculative-Decoding-Typ (llama.cpp) |
|
||||
@@ -0,0 +1,92 @@
|
||||
"""
|
||||
Mission Control 2.0 — dünner FastAPI-Einstieg.
|
||||
|
||||
Hängt die Router ein, liefert (in Prod) das gebaute React-Frontend aus und
|
||||
setzt eine no-cache-Middleware. Im Dev läuft das Frontend über den Vite-Dev-
|
||||
Server (proxyt /api hierher), daher CORS für localhost offen.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
from contextlib import asynccontextmanager
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.responses import FileResponse
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from starlette.requests import Request
|
||||
|
||||
from config import FRONTEND_DIST, VERSION
|
||||
from routers import agent, connect, gateway_proxy, health, maintenance, memory, models, routing, system, voice
|
||||
from services import warmer
|
||||
|
||||
# Zentrales Logging — Level via MC_LOG_LEVEL (INFO default). Eine Konfiguration
|
||||
# für alle Module (logging.getLogger(__name__)).
|
||||
logging.basicConfig(
|
||||
level=os.environ.get("MC_LOG_LEVEL", "INFO").upper(),
|
||||
format="%(asctime)s %(levelname)-7s %(name)s: %(message)s",
|
||||
)
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI):
|
||||
"""Hintergrund-Tasks an den App-Lebenszyklus binden: Re-Warm-Wächter fürs Agent-Hirn."""
|
||||
task = asyncio.create_task(warmer.rewarm_loop()) if warmer.ENABLED else None
|
||||
if task:
|
||||
log.info("Hirn-Re-Warm-Wächter aktiv (Intervall %ss, Hirn dynamisch aus Hermes-Config)", warmer.INTERVAL)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
if task:
|
||||
task.cancel()
|
||||
|
||||
|
||||
app = FastAPI(title="Mission Control 2.0", version=VERSION, lifespan=lifespan)
|
||||
|
||||
# Dev: Vite-Dev-Server (5173) ruft das Backend per /api auf.
|
||||
app.add_middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=["http://localhost:5173", "http://127.0.0.1:5173"],
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
|
||||
@app.middleware("http")
|
||||
async def no_cache(request: Request, call_next):
|
||||
resp = await call_next(request)
|
||||
if request.url.path.startswith("/api"):
|
||||
resp.headers["Cache-Control"] = "no-cache"
|
||||
return resp
|
||||
|
||||
|
||||
app.include_router(health.router)
|
||||
app.include_router(models.router)
|
||||
app.include_router(routing.router)
|
||||
app.include_router(system.router)
|
||||
app.include_router(connect.router)
|
||||
app.include_router(memory.router)
|
||||
app.include_router(agent.router)
|
||||
app.include_router(voice.router) # Sprache: STT/TTS-Proxy + Hermes-Agent-Chat (Voice-Tab)
|
||||
app.include_router(gateway_proxy.router) # OpenAI-kompatibler /v1-Gateway (model:auto)
|
||||
app.include_router(maintenance.router)
|
||||
|
||||
|
||||
# Prod: gebautes Frontend ausliefern (falls vorhanden). SPA-Fallback auf index.html.
|
||||
if FRONTEND_DIST.exists():
|
||||
app.mount("/assets", StaticFiles(directory=FRONTEND_DIST / "assets"), name="assets")
|
||||
|
||||
@app.get("/{full_path:path}")
|
||||
def spa(full_path: str):
|
||||
# Falls die Datei direkt in FRONTEND_DIST liegt (z.B. manifest.webmanifest, favicon.ico), liefere sie aus
|
||||
target = FRONTEND_DIST / full_path
|
||||
if target.is_file():
|
||||
return FileResponse(target)
|
||||
|
||||
index = FRONTEND_DIST / "index.html"
|
||||
if index.exists():
|
||||
# index.html nie cachen → Browser zieht nach jedem Deploy das aktuelle (gehashte) Bundle.
|
||||
return FileResponse(index, headers={"Cache-Control": "no-cache, must-revalidate"})
|
||||
return {"detail": "frontend not built"}
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
"""
|
||||
Zentrale Konfiguration für Mission Control 2.0.
|
||||
|
||||
Eine Quelle der Wahrheit für Pfade, URLs und Defaults — alles über Env-Vars
|
||||
überschreibbar. Bewusst schlank: MC 2.0 ist ein Glue-Cockpit, das vorhandene
|
||||
Dienste (llama-swap, LiteLLM-Gateway, Hermes) steuert, statt sie nachzubauen.
|
||||
"""
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from ruamel.yaml import YAML
|
||||
|
||||
# --- Engine (llama-swap) -----------------------------------------------------
|
||||
LLAMA_SWAP_URL = os.environ.get("MC_LLAMA_SWAP_URL", "http://127.0.0.1:8080").rstrip("/")
|
||||
CONFIG_PATH = Path(os.environ.get("MC_CONFIG_PATH", "/etc/llama-swap/config.yaml"))
|
||||
MODELS_DIR = Path(os.environ.get("MC_MODELS_DIR", "/srv/models"))
|
||||
# Cache der Modell-Entdeckung ("aktuell beste Modelle", live von HuggingFace).
|
||||
# Persistent neben den Modellen (übersteht Deploys). TTL = Frische-Fenster.
|
||||
DISCOVER_CACHE_PATH = Path(os.environ.get("MC_DISCOVER_CACHE", str(MODELS_DIR / "mc2-discover.json")))
|
||||
DISCOVER_TTL = int(os.environ.get("MC_DISCOVER_TTL", "43200")) # 12 h
|
||||
# Geteiltes Gedächtnis (SQLite, WAL). Persistent neben den Modellen.
|
||||
# Hinweis: nur noch für die einmalige Mem0-Migration relevant — das aktive Gedächtnis
|
||||
# liegt jetzt in Mem0/Chroma hinter dem Sidecar (siehe MEM0_SERVICE_URL).
|
||||
MEMORY_DB = Path(os.environ.get("MC_MEMORY_DB", str(MODELS_DIR / "mc2-memory.db")))
|
||||
# Mem0-Sidecar (auto-lernendes, semantisches Gedächtnis). Läuft im ~/.mem0/venv (Python 3.12),
|
||||
# weil mem0+chromadb unter dem 3.14-Backend nicht laufen. MC2 spricht ihn lokal per HTTP an.
|
||||
MEM0_SERVICE_URL = os.environ.get("MC_MEM0_SERVICE_URL", "http://127.0.0.1:8765").rstrip("/")
|
||||
# Befehl-Vorlage für llama-swap: {model}=GGUF-Pfad, {ctx}=Kontext, ${PORT} bleibt stehen.
|
||||
# Hinweis: --prompt-cache/--prompt-cache-all sind llama-CLI-Flags, NICHT llama-server —
|
||||
# llama-server lehnt sie ab ("invalid argument") und startet dann nicht. Prompt-Caching
|
||||
# macht llama-server ohnehin automatisch pro Slot (KV-Reuse).
|
||||
_DEFAULT_CMD_TEMPLATE = (
|
||||
"llama-server -m {model} --host 127.0.0.1 --port ${PORT} "
|
||||
"-c {ctx} -ngl 999 -fa on --no-mmap"
|
||||
)
|
||||
CMD_TEMPLATE = os.environ.get("MC_CMD_TEMPLATE", _DEFAULT_CMD_TEMPLATE)
|
||||
if "{model}" not in CMD_TEMPLATE:
|
||||
CMD_TEMPLATE = _DEFAULT_CMD_TEMPLATE
|
||||
DEFAULT_TTL = int(os.environ.get("MC_DEFAULT_TTL", "300"))
|
||||
# Verzeichnis mit Draft-Modellen für Speculative Decoding. Beim Hinzufügen eines
|
||||
# fast/coder-Modells wird hieraus automatisch ein **vocab-kompatibler** Draft gewählt
|
||||
# (Vocab-Check via services.gguf_meta; ein inkompatibler Draft lässt llama.cpp scheitern).
|
||||
DRAFTS_DIR = Path(os.environ.get("MC_DRAFTS_DIR", str(MODELS_DIR / "drafts")))
|
||||
# Optionaler expliziter Default-Draft (leer = Auto-Erkennung aus DRAFTS_DIR). Wird nur
|
||||
# verwendet, wenn er zum Ziel-Modell vocab-kompatibel ist. (Früher fix qwen2.5 → entfernt,
|
||||
# weil das mit neueren Vocabs wie Qwen3.6 inkompatibel ist und Spec stillschweigend brach.)
|
||||
SPEC_DRAFT_MODEL_PATH = os.environ.get("MC_SPEC_DRAFT_MODEL", "")
|
||||
# Speculative-Decoding-Typ (llama.cpp dieser Generation braucht --spec-type zusätzlich
|
||||
# zu --spec-draft-model, sonst ist Spec inaktiv).
|
||||
SPEC_TYPE = os.environ.get("MC_SPEC_TYPE", "draft-simple")
|
||||
# MTP-Speculative-Decoding (Multi-Token-Prediction): manche Modelle bringen einen eigenen
|
||||
# MTP-Kopf mit (z.B. gemma-4 → arch 'gemma4-assistant', Datei 'mtp-*.gguf'). Der wird mit
|
||||
# `--model-draft <mtp.gguf> --spec-type draft-mtp --spec-draft-n-max N` geladen (NICHT
|
||||
# --spec-draft-model/draft-simple). 1,5–2× Durchsatz bei null Qualitätsverlust.
|
||||
SPEC_DRAFT_N_MAX = int(os.environ.get("MC_SPEC_DRAFT_N_MAX", "4"))
|
||||
# Env für HuggingFace-Downloads: XET deaktivieren (Hänger bei ~6 MB, siehe v1-Gotcha).
|
||||
HF_DOWNLOAD_ENV = {"HF_HUB_DISABLE_XET": "1"}
|
||||
|
||||
# --- Routing-Gateway (builtin in MC2, model: auto) ---------------------------
|
||||
# MC2 IST der Gateway (services/gateway.py + routers/gateway_proxy.py). KEIN externer
|
||||
# LiteLLM-Dienst (scheitert auf Python 3.14). Daher keine Gateway-Config-Datei mehr.
|
||||
GATEWAY_URL = os.environ.get("MC_GATEWAY_URL", f"http://127.0.0.1:{os.environ.get('MC_PORT', '9000')}").rstrip("/")
|
||||
|
||||
# --- Hermes Agent (eigener Dienst auf der Box) -------------------------------
|
||||
# Gateway (OpenAI-API des Agenten) + interaktives Web-Terminal (ttyd → `hermes chat`).
|
||||
HERMES_API_URL = os.environ.get("HERMES_API_URL", "http://127.0.0.1:8642").rstrip("/")
|
||||
# API-Key der Hermes-`api_server`-Plattform (~/.hermes/.env: API_SERVER_KEY). Nötig für
|
||||
# /v1/chat/completions (Voice-Pipeline) — Bearer-Auth, sonst 401. Derselbe volle Agent
|
||||
# (Tools + geteiltes Mem0) wie CLI/Telegram, nur über HTTP.
|
||||
def _read_hermes_env(key: str) -> str:
|
||||
"""Liest einen Schlüssel aus ~/.hermes/.env (Fallback, falls nicht in der Prozess-Env).
|
||||
Der MC2-Dienst erbt die Hermes-Secrets sonst nicht."""
|
||||
try:
|
||||
env_path = Path(os.path.expanduser(os.environ.get("HERMES_HOME", "~/.hermes"))) / ".env"
|
||||
for line in env_path.read_text(encoding="utf-8").splitlines():
|
||||
line = line.strip()
|
||||
if line.startswith(f"{key}="):
|
||||
return line.split("=", 1)[1].strip().strip('"').strip("'")
|
||||
except OSError:
|
||||
pass
|
||||
return ""
|
||||
|
||||
|
||||
HERMES_API_KEY = (
|
||||
os.environ.get("HERMES_API_KEY")
|
||||
or os.environ.get("API_SERVER_KEY")
|
||||
or _read_hermes_env("API_SERVER_KEY")
|
||||
)
|
||||
# Modellfeld im OpenAI-Request; die api_server-Plattform nutzt ihr konfiguriertes Hirn,
|
||||
# das Feld ist i.d.R. kosmetisch. Override via Env, falls die Plattform strikt prüft.
|
||||
HERMES_API_MODEL = os.environ.get("HERMES_API_MODEL", "hermes")
|
||||
|
||||
# --- Voice-Sidecar (STT faster-whisper + TTS Piper/Chatterbox) ---------------
|
||||
# Eigenes Python-3.12-venv (~/.voice/venv), analog Mem0-Sidecar. MC2 proxyt nach außen.
|
||||
VOICE_SERVICE_URL = os.environ.get("MC_VOICE_SERVICE_URL", "http://127.0.0.1:8650").rstrip("/")
|
||||
# Hermes-Terminal: ttyd-Web-Terminal der interaktiven Agent-CLI (Ersatz für AnythingLLM-Chat).
|
||||
# Wird in MC2 per iframe eingebettet (Terminal-Seite). Siehe deploy/hermes-terminal.service.
|
||||
HERMES_TERMINAL_URL = os.environ.get("MC_HERMES_TERMINAL_URL", "http://192.168.178.151:7681").rstrip("/")
|
||||
# GitHub-Repo für Update-Checks.
|
||||
HERMES_AGENT_REPO = os.environ.get("MC_HERMES_AGENT_REPO", "NousResearch/hermes-agent")
|
||||
HERMES_HOME = Path(os.path.expanduser(os.environ.get("HERMES_HOME", "~/.hermes")))
|
||||
# PC Executor — läuft auf dem Windows-PC, erreichbar über LAN.
|
||||
PC_EXECUTOR_URL = os.environ.get("MC_PC_EXECUTOR_URL", "http://192.168.178.98:7777").rstrip("/")
|
||||
|
||||
# --- Server ------------------------------------------------------------------
|
||||
HOST = os.environ.get("MC_HOST", "0.0.0.0")
|
||||
PORT = int(os.environ.get("MC_PORT", "9000"))
|
||||
# Gebautes React-Frontend (frontend/dist). In Prod liefert FastAPI es statisch aus;
|
||||
# im Dev läuft der Vite-Dev-Server separat und proxyt /api hierher.
|
||||
FRONTEND_DIST = Path(os.environ.get("MC_FRONTEND_DIST", str(Path(__file__).resolve().parent.parent / "frontend" / "dist")))
|
||||
|
||||
# Version (Phase 0 — Greenfield-Skeleton).
|
||||
VERSION = "2.0.0-w8"
|
||||
|
||||
# Gemeinsame YAML-Instanz (preserve_quotes hält Kommentare/Quotes in config.yaml).
|
||||
yaml = YAML()
|
||||
yaml.preserve_quotes = True
|
||||
@@ -0,0 +1,56 @@
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
# Add backend directory to sys.path so we can import services
|
||||
sys.path.append(str(Path(__file__).resolve().parent))
|
||||
|
||||
from services.llamaswap import read_config, write_config, spec_draft_flags, _PATH_RE
|
||||
from config import CONFIG_PATH
|
||||
|
||||
def migrate():
|
||||
print(f"Reading config from {CONFIG_PATH}...")
|
||||
if not CONFIG_PATH.exists():
|
||||
print(f"Config path {CONFIG_PATH} does not exist. Skipping.")
|
||||
return
|
||||
|
||||
cfg = read_config()
|
||||
models = cfg.get("models", {})
|
||||
|
||||
for name, spec in models.items():
|
||||
if not isinstance(spec, dict):
|
||||
continue
|
||||
cmd = spec.get("cmd", "")
|
||||
if not cmd:
|
||||
continue
|
||||
|
||||
print(f"Migrating model: {name}")
|
||||
|
||||
# 1. Defektes --prompt-cache/--prompt-cache-all entfernen (llama-CLI-Flags,
|
||||
# die llama-server ablehnt → Start scheitert). Caching macht llama-server
|
||||
# automatisch pro Slot.
|
||||
cmd = cmd.replace(" --prompt-cache-all", "").replace(" --prompt-cache", "")
|
||||
|
||||
# 2. Extract aliases/role
|
||||
aliases = spec.get("aliases", [])
|
||||
role = aliases[0] if aliases else None
|
||||
|
||||
# 3. Add parallel + (nur vocab-kompatibles) Speculative Decoding für fast/coder.
|
||||
# spec_draft_flags() prüft die Vocab-Kompatibilität und hängt --spec-type an;
|
||||
# ein inkompatibler Draft (z.B. qwen2.5 ↔ Qwen3.6) wird NICHT gesetzt.
|
||||
if role in ("fast", "coder"):
|
||||
if "--parallel" not in cmd:
|
||||
cmd = cmd.strip() + " --parallel 2"
|
||||
if "--spec-draft-model" not in cmd:
|
||||
target = mt.group(1) if (mt := _PATH_RE.search(cmd)) else ""
|
||||
cmd = cmd.strip() + spec_draft_flags(target)
|
||||
|
||||
# Update cmd
|
||||
from ruamel.yaml.scalarstring import LiteralScalarString
|
||||
spec["cmd"] = LiteralScalarString(cmd.strip() + "\n")
|
||||
|
||||
print(f"Writing updated config back to {CONFIG_PATH}...")
|
||||
write_config(cfg)
|
||||
print("Migration completed successfully!")
|
||||
|
||||
if __name__ == "__main__":
|
||||
migrate()
|
||||
@@ -0,0 +1,60 @@
|
||||
{
|
||||
"_comment": "Kuratierter Modell-Katalog (Cookbook) für Strix Halo / Ryzen AI MAX+ 395 — 128GB unified, bandbreiten-limitiert (256 GB/s). MoE-first. EINE Quelle der Wahrheit für KORREKTE Metadaten (total/active params, moe, generation) → präzise Empfehlungen ohne Namens-Raterei. Inspiriert vom Odysseus-Cookbook (statischer, validierter Katalog statt Live-Scraping). Erweiterbar: neue Modelle hier eintragen. Felder: name (Match-Identifier), repo (HF org/name für Install), family (+Subtyp), generation (numerisch, für Upgrade-Vergleich), total_params_b, active_params_b (=total bei dense), moe, quant, ctx (empfohlen), tools, vision.",
|
||||
"version": "2026-06-27",
|
||||
"models": [
|
||||
{
|
||||
"role": "fast", "name": "Qwen3.6-35B-A3B", "repo": "Qwen/Qwen3.6-35B-A3B-GGUF",
|
||||
"family": "qwen", "generation": 3.6, "total_params_b": 35, "active_params_b": 3,
|
||||
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": true
|
||||
},
|
||||
{
|
||||
"role": "fast", "name": "Qwen3-30B-A3B-Instruct", "repo": "unsloth/Qwen3-30B-A3B-Instruct-2507-GGUF",
|
||||
"family": "qwen", "generation": 3.0, "total_params_b": 30, "active_params_b": 3,
|
||||
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": false
|
||||
},
|
||||
|
||||
{
|
||||
"role": "heavy", "name": "Qwen3.5-122B-A10B", "repo": "Qwen/Qwen3.5-122B-A10B-GGUF",
|
||||
"family": "qwen", "generation": 3.5, "total_params_b": 122, "active_params_b": 10,
|
||||
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": false
|
||||
},
|
||||
{
|
||||
"role": "heavy", "name": "gpt-oss-120b", "repo": "ggml-org/gpt-oss-120b-GGUF",
|
||||
"family": "gpt-oss", "generation": 1.0, "total_params_b": 120, "active_params_b": 5,
|
||||
"moe": true, "quant": "MXFP4", "ctx": 32768, "tools": true, "vision": false
|
||||
},
|
||||
|
||||
{
|
||||
"role": "coder", "name": "Qwen3-Coder-Next", "repo": "Qwen/Qwen3-Coder-Next-GGUF",
|
||||
"family": "qwen-coder", "generation": 3.0, "total_params_b": 84, "active_params_b": 3,
|
||||
"moe": true, "quant": "Q4_K_M", "ctx": 65536, "tools": true, "vision": false
|
||||
},
|
||||
{
|
||||
"role": "coder", "name": "Qwen3-Coder-30B-A3B-Instruct", "repo": "unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF",
|
||||
"family": "qwen-coder", "generation": 3.0, "total_params_b": 30, "active_params_b": 3,
|
||||
"moe": true, "quant": "Q4_K_M", "ctx": 65536, "tools": true, "vision": false
|
||||
},
|
||||
|
||||
{
|
||||
"role": "vision", "name": "Qwen3-VL-8B-Instruct", "repo": "Qwen/Qwen3-VL-8B-Instruct-GGUF",
|
||||
"family": "qwen-vl", "generation": 3.0, "total_params_b": 8, "active_params_b": 8,
|
||||
"moe": false, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
|
||||
},
|
||||
{
|
||||
"role": "vision", "name": "Qwen3-VL-2B-Instruct", "repo": "Qwen/Qwen3-VL-2B-Instruct-GGUF",
|
||||
"family": "qwen-vl", "generation": 3.0, "total_params_b": 2, "active_params_b": 2,
|
||||
"moe": false, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
|
||||
},
|
||||
|
||||
{
|
||||
"role": "scout", "name": "gemma-4-26B-A4B-it", "repo": "google/gemma-4-26B-A4B-it-GGUF",
|
||||
"family": "gemma", "generation": 4.0, "total_params_b": 26, "active_params_b": 4,
|
||||
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
|
||||
},
|
||||
{
|
||||
"role": "scout", "name": "gemma-4-31B-it", "repo": "google/gemma-4-31B-it-GGUF",
|
||||
"family": "gemma", "generation": 4.0, "total_params_b": 31, "active_params_b": 31,
|
||||
"moe": false, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
fastapi>=0.115
|
||||
uvicorn[standard]>=0.30
|
||||
httpx>=0.27
|
||||
ruamel.yaml>=0.18
|
||||
psutil>=5.9
|
||||
huggingface_hub>=0.27
|
||||
mcp>=1.2.0
|
||||
@@ -0,0 +1,44 @@
|
||||
"""Agent-Endpoint: Hermes-Status + WebUI-Link (MC verlinkt nur, betreibt nicht)."""
|
||||
|
||||
from fastapi import APIRouter
|
||||
from pydantic import BaseModel
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
from services.agent import agent_status, hermes_brain_info, set_agent_brain, update_brain_model
|
||||
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
|
||||
class BrainReq(BaseModel):
|
||||
model: str
|
||||
|
||||
|
||||
class SetBrainReq(BaseModel):
|
||||
model_id: str
|
||||
|
||||
|
||||
@router.get("/agent/status")
|
||||
def status() -> dict:
|
||||
return agent_status()
|
||||
|
||||
|
||||
@router.get("/agent/brain")
|
||||
def brain_info() -> dict:
|
||||
"""Aktuelles Agent-Hirn (hermes) + bestes NousResearch-Hermes-Update."""
|
||||
return hermes_brain_info()
|
||||
|
||||
|
||||
@router.post("/agent/brain/set")
|
||||
def set_brain(body: SetBrainReq) -> dict:
|
||||
"""Setzt ein installiertes Modell als Agent-Hirn (Alias + warm-Gruppe + Config)."""
|
||||
res = set_agent_brain(body.model_id)
|
||||
if not res.get("ok"):
|
||||
raise HTTPException(400, res.get("reason", "Fehler beim Setzen des Agent-Hirns"))
|
||||
return res
|
||||
|
||||
|
||||
@router.post("/agent/brain")
|
||||
def set_brain_model(body: BrainReq) -> dict:
|
||||
ok = update_brain_model(body.model)
|
||||
return {"ok": ok}
|
||||
@@ -0,0 +1,21 @@
|
||||
"""Connect-Endpoint: erzeugt IDE-/Agent-Snippets (auf den Gateway + Memory-MCP)."""
|
||||
|
||||
from fastapi import APIRouter
|
||||
|
||||
from services.connect import DEFAULT_HOST, build_snippets, check_health
|
||||
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
|
||||
@router.get("/connect")
|
||||
def connect(host: str = DEFAULT_HOST, mcp_path: str | None = None) -> dict:
|
||||
kwargs = {}
|
||||
if mcp_path:
|
||||
kwargs["mcp_script_path"] = mcp_path
|
||||
return build_snippets(host=host, **kwargs)
|
||||
|
||||
|
||||
@router.get("/connect/health")
|
||||
def connect_health() -> dict:
|
||||
"""Live-Status der zwei Leitungen (Gateway + Gedächtnis) für den Verbinden-Tab."""
|
||||
return check_health()
|
||||
@@ -0,0 +1,69 @@
|
||||
import httpx
|
||||
from fastapi import APIRouter, Request
|
||||
from fastapi.responses import JSONResponse, StreamingResponse
|
||||
|
||||
from config import LLAMA_SWAP_URL
|
||||
from services.gateway_stream import record_stream_chunk, record_usage
|
||||
from services.router_logic import LANES, choose_for_lane
|
||||
from services.routing_policy import load_policy
|
||||
|
||||
router = APIRouter(prefix="/v1")
|
||||
|
||||
# Virtuelle Lanes, die der Gateway zusätzlich zu den echten Modellen als „Modell" anbietet.
|
||||
_LANE_LABELS = {"coding": "Coding (Router → coder/heavy/fast)", "chat": "Chat (Router → fast/heavy)"}
|
||||
|
||||
|
||||
@router.get("/models")
|
||||
async def models():
|
||||
async with httpx.AsyncClient(timeout=10) as c:
|
||||
r = await c.get(f"{LLAMA_SWAP_URL}/v1/models")
|
||||
data = r.json()
|
||||
# Lanes ganz oben einblenden, damit IDEs einfach „coding"/„chat" wählen können.
|
||||
lanes = [{"id": lane, "object": "model", "owned_by": "mc2-router",
|
||||
"description": _LANE_LABELS.get(lane, lane)} for lane in LANES]
|
||||
if isinstance(data, dict) and isinstance(data.get("data"), list):
|
||||
data["data"] = lanes + data["data"]
|
||||
return JSONResponse(data, status_code=r.status_code)
|
||||
|
||||
|
||||
async def _proxy(path: str, request: Request):
|
||||
body = await request.json()
|
||||
requested = str(body.get("model") or "auto")
|
||||
if requested.lower() in ("auto", "chat", "coding"):
|
||||
lane = requested.lower()
|
||||
alias, reason = choose_for_lane(lane, body)
|
||||
body["model"] = alias
|
||||
routed = {"x-mc-routed-to": alias, "x-mc-route-reason": reason, "x-mc-lane": lane}
|
||||
else:
|
||||
alias = requested
|
||||
routed = {"x-mc-routed-to": requested}
|
||||
# fast-Spur: Thinking aus für flotte Antworten (sofern Client es nicht selbst setzt).
|
||||
pol = load_policy()
|
||||
if pol["fast_no_think"] and alias == pol["fast"] and "chat_template_kwargs" not in body:
|
||||
body["chat_template_kwargs"] = {"enable_thinking": False}
|
||||
url = f"{LLAMA_SWAP_URL}{path}"
|
||||
|
||||
if body.get("stream"):
|
||||
async def gen():
|
||||
async with httpx.AsyncClient(timeout=None) as c:
|
||||
async with c.stream("POST", url, json=body) as r:
|
||||
async for chunk in r.aiter_raw():
|
||||
record_stream_chunk(chunk, alias)
|
||||
yield chunk
|
||||
return StreamingResponse(gen(), media_type="text/event-stream", headers=routed)
|
||||
|
||||
async with httpx.AsyncClient(timeout=600) as c:
|
||||
r = await c.post(url, json=body)
|
||||
resp_json = r.json()
|
||||
record_usage(resp_json.get("usage") if isinstance(resp_json, dict) else None, alias)
|
||||
return JSONResponse(resp_json, status_code=r.status_code, headers=routed)
|
||||
|
||||
|
||||
@router.post("/chat/completions")
|
||||
async def chat_completions(request: Request):
|
||||
return await _proxy("/v1/chat/completions", request)
|
||||
|
||||
|
||||
@router.post("/completions")
|
||||
async def completions(request: Request):
|
||||
return await _proxy("/v1/completions", request)
|
||||
@@ -0,0 +1,21 @@
|
||||
"""Health-/Status-Endpoint — schlanker Lebenszeichen-Check für MC 2.0."""
|
||||
|
||||
from fastapi import APIRouter
|
||||
|
||||
from config import VERSION
|
||||
from services import gateway, llamaswap
|
||||
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
|
||||
@router.get("/health")
|
||||
def health() -> dict:
|
||||
return {
|
||||
"status": "ok",
|
||||
"version": VERSION,
|
||||
"engine_reachable": llamaswap.engine_reachable(),
|
||||
"gateway_reachable": gateway.gateway_reachable(),
|
||||
# Echte Hirn-Bereitschaft: Engine kann erreichbar sein, das Agent-Hirn ('fast') aber tot
|
||||
# (Crash/OOM nach Engine-Update). Das wäre sonst ein silent fail (App-Fehler statt Status).
|
||||
"brain": llamaswap.brain_status(),
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
"""Wartungs-Endpoints: Update-Badge, OS-/Engine-Update, Reboot, Restart, Logs."""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Header
|
||||
from pydantic import BaseModel
|
||||
|
||||
from services import maintenance
|
||||
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
|
||||
class SudoReq(BaseModel):
|
||||
sudo_password: str | None = None
|
||||
|
||||
|
||||
class RestartReq(BaseModel):
|
||||
service: str
|
||||
sudo_password: str | None = None
|
||||
|
||||
|
||||
@router.get("/maintenance/updates")
|
||||
def updates() -> dict:
|
||||
return maintenance.updates()
|
||||
|
||||
|
||||
@router.get("/maintenance/update-details")
|
||||
def update_details(kind: str) -> dict:
|
||||
if kind not in ("os", "engine", "swap", "hermes"):
|
||||
raise HTTPException(400, "Unbekannte Update-Art.")
|
||||
return maintenance.update_details(kind)
|
||||
|
||||
@router.post("/maintenance/check-updates")
|
||||
def check_updates(body: SudoReq) -> dict:
|
||||
res = maintenance.check_updates_job(body.sudo_password)
|
||||
if isinstance(res, dict) and not res.get("ok", True):
|
||||
return res
|
||||
return res
|
||||
|
||||
|
||||
@router.post("/maintenance/os-update")
|
||||
def os_update(body: SudoReq) -> dict:
|
||||
res = maintenance.os_update_job(body.sudo_password)
|
||||
if isinstance(res, dict) and not res.get("ok", True):
|
||||
return res
|
||||
return res
|
||||
|
||||
|
||||
@router.post("/maintenance/engine-update")
|
||||
def engine_update(body: SudoReq) -> dict:
|
||||
res = maintenance.engine_update_job(body.sudo_password)
|
||||
if not res:
|
||||
raise HTTPException(400, "Kein Engine-Update-Befehl gesetzt (MC_ENGINE_UPDATE_CMD).")
|
||||
return res
|
||||
|
||||
|
||||
@router.post("/maintenance/swap-update")
|
||||
def swap_update(body: SudoReq) -> dict:
|
||||
res = maintenance.swap_update_job(body.sudo_password)
|
||||
if not res:
|
||||
raise HTTPException(400, "Kein Router-Update-Befehl gesetzt (MC_SWAP_UPDATE_CMD).")
|
||||
return res
|
||||
|
||||
|
||||
@router.post("/maintenance/hermes-update")
|
||||
def hermes_update() -> dict:
|
||||
return maintenance.hermes_update_job()
|
||||
|
||||
|
||||
@router.post("/maintenance/reboot")
|
||||
def reboot(body: SudoReq) -> dict:
|
||||
return maintenance.reboot(body.sudo_password)
|
||||
|
||||
|
||||
@router.post("/maintenance/restart")
|
||||
def restart(body: RestartReq) -> dict:
|
||||
return maintenance.restart_service(body.service, body.sudo_password)
|
||||
|
||||
|
||||
@router.get("/maintenance/logs")
|
||||
def logs(service: str, lines: int = 200, x_sudo_password: str | None = Header(None)) -> dict:
|
||||
return maintenance.logs(service, lines, x_sudo_password)
|
||||
@@ -0,0 +1,81 @@
|
||||
"""Memory-Endpoints (geteiltes Gedächtnis). LAN-only, kein Token in 2.0-Phase 3."""
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from services import memory
|
||||
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
|
||||
class MemIn(BaseModel):
|
||||
content: str
|
||||
category: str = "knowledge"
|
||||
source: str = "manual"
|
||||
|
||||
|
||||
class MemUp(BaseModel):
|
||||
content: str | None = None
|
||||
category: str | None = None
|
||||
|
||||
|
||||
class DedupeIn(BaseModel):
|
||||
apply: bool = False
|
||||
threshold: float = 0.85
|
||||
|
||||
|
||||
class LearnIn(BaseModel):
|
||||
text: str | None = None
|
||||
messages: list[dict] | None = None
|
||||
source: str = "auto"
|
||||
category: str = "knowledge"
|
||||
|
||||
|
||||
@router.get("/memory/export")
|
||||
def export() -> dict:
|
||||
return memory.export_text()
|
||||
|
||||
|
||||
@router.get("/memory/graph")
|
||||
def graph(min_score: float = 0.45, top_k: int = 3) -> dict:
|
||||
"""Fakten als Ähnlichkeits-Graph (Knoten + semantische Kanten) für die Visualisierung."""
|
||||
return memory.graph(min_score=min_score, top_k=top_k)
|
||||
|
||||
|
||||
@router.post("/memory/learn", status_code=201)
|
||||
def learn(body: LearnIn) -> dict:
|
||||
"""Auto-Lernen: Gesprächs-Turns/Text durchreichen → Mem0 extrahiert Fakten selbst."""
|
||||
return memory.learn(text=body.text, messages=body.messages,
|
||||
source=body.source, category=body.category)
|
||||
|
||||
|
||||
@router.post("/memory/dedupe")
|
||||
def dedupe(body: DedupeIn) -> dict:
|
||||
return memory.dedupe(apply=body.apply, threshold=body.threshold)
|
||||
|
||||
|
||||
@router.get("/memory")
|
||||
def list_mem(q: str = "", category: str = "") -> list[dict]:
|
||||
return memory.list_memories(q=q, category=category)
|
||||
|
||||
|
||||
@router.post("/memory", status_code=201)
|
||||
def add(body: MemIn) -> dict:
|
||||
if body.category not in memory.CATEGORIES:
|
||||
raise HTTPException(400, f"Kategorie '{body.category}' unbekannt.")
|
||||
return memory.add_memory(body.content, body.category, body.source)
|
||||
|
||||
|
||||
@router.put("/memory/{mid}")
|
||||
def update(mid: str, body: MemUp) -> dict:
|
||||
res = memory.update_memory(mid, content=body.content, category=body.category)
|
||||
if not res:
|
||||
raise HTTPException(404, "Eintrag nicht gefunden")
|
||||
return res
|
||||
|
||||
|
||||
@router.delete("/memory/{mid}")
|
||||
def delete(mid: str) -> dict:
|
||||
if not memory.delete_memory(mid):
|
||||
raise HTTPException(404, "Eintrag nicht gefunden")
|
||||
return {"ok": True}
|
||||
@@ -0,0 +1,293 @@
|
||||
"""Modelle-Endpoints: Liste (mit Caps), Discover, Fit, Register, Groups."""
|
||||
|
||||
import psutil
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from config import HF_DOWNLOAD_ENV, MODELS_DIR
|
||||
from services import budget, discover, hf, jobengine, llamaswap
|
||||
from services.fit import evaluate_fit, max_ctx_for
|
||||
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
|
||||
def _ram_gb() -> float:
|
||||
return psutil.virtual_memory().total / (1024 ** 3)
|
||||
|
||||
|
||||
@router.get("/models")
|
||||
def models() -> dict:
|
||||
items = llamaswap.list_models()
|
||||
return {"models": items, "count": len(items), "running": llamaswap.get_running_models()}
|
||||
|
||||
|
||||
@router.get("/discover")
|
||||
def discover_models(force: bool = False) -> dict:
|
||||
ram = _ram_gb()
|
||||
data = discover.refresh_discover(ram) if force else discover.safe_discover(ram)
|
||||
if not data:
|
||||
raise HTTPException(502, "Modell-Quellen gerade nicht erreichbar — später erneut.")
|
||||
return {**data, "sys_ram_gb": round(ram, 1)}
|
||||
|
||||
|
||||
@router.get("/fit")
|
||||
def fit(params_b: float = 0, quant: str = "Q4_K_M", ctx: int = 8192,
|
||||
name: str = "", role: str = "") -> dict:
|
||||
"""Hardware-Fit-Vorschau. params_b<=0 → aus KATALOG (echte Metadaten, MoE-bewusst)
|
||||
oder sonst aus dem Namen geschätzt. assigned_ctx = der ctx, der TATSÄCHLICH vergeben
|
||||
würde: SETUP-BEWUSST (neben Hirn/warmem Set), nicht nur gegen den Gesamt-RAM.
|
||||
So sieht die 'Erweiterte Ansicht' vor dem Download Ampel + echten ctx."""
|
||||
ram = _ram_gb()
|
||||
pb = params_b if params_b > 0 else budget.params_b_for(name)
|
||||
saw = budget.setup_aware_ctx(pb, quant, role=role or None)
|
||||
return {
|
||||
"params_b": round(pb, 1),
|
||||
"fit": evaluate_fit(pb, quant, ctx, ram, name=name),
|
||||
"optimal_ctx": max_ctx_for(pb, quant, ram), # Roh-Obergrenze (Modell allein)
|
||||
"assigned_ctx": saw["ctx"], # setup-bewusst vergeben
|
||||
"budget": {"gtt_gb": saw["gtt_gb"], "reserved_gb": saw["reserved_gb"],
|
||||
"budget_gb": saw["budget_gb"], "mode": saw["mode"]},
|
||||
"sys_ram_gb": round(ram, 1),
|
||||
}
|
||||
|
||||
|
||||
class RegisterReq(BaseModel):
|
||||
model_path: str
|
||||
role: str | None = None
|
||||
ctx: int = 8192
|
||||
ttl: int | None = None
|
||||
mmproj_path: str | None = None
|
||||
jinja: bool = False
|
||||
|
||||
|
||||
@router.post("/models/register")
|
||||
def register(req: RegisterReq) -> dict:
|
||||
try:
|
||||
model_id = llamaswap.register_model(
|
||||
req.model_path, role=req.role, ctx=req.ctx, ttl=req.ttl,
|
||||
mmproj_path=req.mmproj_path, jinja=req.jinja,
|
||||
)
|
||||
except PermissionError as exc:
|
||||
raise HTTPException(500, str(exc))
|
||||
return {"ok": True, "model_id": model_id}
|
||||
|
||||
|
||||
class InstallReq(BaseModel):
|
||||
repo: str
|
||||
role: str | None = None
|
||||
quant: str = "Q4_K_M"
|
||||
ctx: int | None = None
|
||||
jinja: bool = False
|
||||
hf_token: str | None = None
|
||||
|
||||
|
||||
@router.get("/hf/search")
|
||||
def hf_search(q: str = "") -> dict:
|
||||
return {"results": hf.search(q)}
|
||||
|
||||
|
||||
@router.get("/hf/quants")
|
||||
def hf_quants(repo: str) -> dict:
|
||||
repo = hf.normalize_repo(repo)
|
||||
return {"repo": repo, "quants": hf.list_quants(repo)}
|
||||
|
||||
|
||||
@router.post("/models/install")
|
||||
def install(req: InstallReq) -> dict:
|
||||
"""Lädt ein Modell von HuggingFace (Hintergrund-Job) UND trägt es sofort in
|
||||
llama-swap ein (cmd + Rolle-Alias). llama-swap (-watch-config) lädt es, sobald
|
||||
die Datei da ist. Split-GGUFs werden komplett geladen, registriert wird der
|
||||
erste Teil (-00001-of-…). Akzeptiert volle HF-URL ODER org/repo."""
|
||||
repo = hf.normalize_repo(req.repo)
|
||||
info = hf.resolve_gguf(repo, req.quant)
|
||||
if not info["first"]:
|
||||
raise HTTPException(404, f"Keine GGUF-Datei für Quant '{req.quant}' in {repo} gefunden.")
|
||||
|
||||
subdir = repo.split("/")[-1]
|
||||
target = MODELS_DIR / subdir
|
||||
target.mkdir(parents=True, exist_ok=True)
|
||||
model_path = str(target / info["first"])
|
||||
mmproj_path = str(target / info["mmproj"]) if info["mmproj"] else None
|
||||
|
||||
ctx = req.ctx
|
||||
if ctx is None:
|
||||
# SETUP-BEWUSST: größter ctx, der neben Hirn/warmem Set passt (nicht nur Modell allein).
|
||||
ctx = budget.setup_aware_ctx(budget.params_b_for(repo), req.quant, role=req.role)["ctx"]
|
||||
|
||||
# Sofort registrieren (Datei kommt gleich) — robust gegen -watch-config.
|
||||
try:
|
||||
model_id = llamaswap.register_model(
|
||||
model_path, role=req.role, ctx=ctx, mmproj_path=mmproj_path, jinja=req.jinja)
|
||||
except PermissionError as exc:
|
||||
raise HTTPException(500, str(exc))
|
||||
|
||||
# Download-Job: alle GGUF-Teile (+ mmproj) per --include holen.
|
||||
args = [hf.hf_bin(), "download", repo]
|
||||
for f in info["files"]:
|
||||
args.append(f)
|
||||
if info["mmproj"]:
|
||||
args.append(info["mmproj"])
|
||||
args += ["--local-dir", str(target)]
|
||||
env = dict(HF_DOWNLOAD_ENV)
|
||||
if req.hf_token:
|
||||
env["HF_TOKEN"] = req.hf_token
|
||||
job_id = jobengine.start_job(args, f"download {req.repo}", env=env)
|
||||
jobengine.attach_download_progress(job_id, str(target), info["total_bytes"])
|
||||
return {"ok": True, "job_id": job_id, "model_id": model_id, "model_path": model_path,
|
||||
"total_bytes": info["total_bytes"], "files": len(info["files"])}
|
||||
|
||||
|
||||
@router.get("/jobs")
|
||||
def jobs() -> dict:
|
||||
return {"jobs": jobengine.public_jobs()}
|
||||
|
||||
|
||||
@router.post("/jobs/{job_id}/cancel")
|
||||
def cancel(job_id: str) -> dict:
|
||||
return {"ok": jobengine.cancel_job(job_id)}
|
||||
|
||||
|
||||
class RoleReq(BaseModel):
|
||||
role: str | None = None
|
||||
|
||||
|
||||
@router.get("/roles/{role}/recommend")
|
||||
def recommend_role(role: str) -> dict:
|
||||
"""Welches installierte Modell passt am besten auf diese Rolle? (Capability + setup-
|
||||
bewusster Fit). Basis für 'Empfohlen'-Hinweis + Auto-Pick im Rollen-Zuweisungs-Modal."""
|
||||
from services import roles
|
||||
return roles.recommend_for_role(role)
|
||||
|
||||
|
||||
@router.post("/models/{model_id}/role")
|
||||
def set_model_role(model_id: str, body: RoleReq) -> dict:
|
||||
# Das Agent-Hirn (Rolle 'hermes') braucht den warm-bewussten Flow (Alias + brains-Gruppe +
|
||||
# ttl 0 + Hermes config.default + Gateway-Restart) — Single Source of Truth UI ↔ Hermes.
|
||||
if (body.role or "").strip().lower() == "hermes":
|
||||
from services.agent import set_agent_brain
|
||||
res = set_agent_brain(model_id)
|
||||
if not res.get("ok"):
|
||||
raise HTTPException(400, res.get("reason", "Fehler beim Setzen des Agent-Hirns"))
|
||||
return res
|
||||
if not llamaswap.set_role(model_id, body.role):
|
||||
raise HTTPException(404, "Modell nicht gefunden")
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
class CtxReq(BaseModel):
|
||||
ctx: int
|
||||
|
||||
|
||||
@router.get("/models/{model_id}/ctx/auto")
|
||||
def auto_ctx(model_id: str) -> dict:
|
||||
"""Setup-bewusster Optimal-ctx für ein bestehendes Modell (Rolle/Params/Quant +
|
||||
aktuelles Setup). Basis für den 'Auto'-Button an der Modellkarte."""
|
||||
m = next((x for x in llamaswap.list_models() if x["name"] == model_id), None)
|
||||
if not m:
|
||||
raise HTTPException(404, "Modell nicht gefunden")
|
||||
saw = budget.setup_aware_ctx_for_model(m)
|
||||
return {"model_id": model_id, "current_ctx": m.get("ctx"),
|
||||
"params_b": round(budget.params_of_model(m), 1), "quant": m.get("quant"),
|
||||
"role": m.get("role"), **saw}
|
||||
|
||||
|
||||
@router.post("/models/{model_id}/ctx")
|
||||
def set_model_ctx(model_id: str, body: CtxReq) -> dict:
|
||||
if not llamaswap.set_ctx(model_id, body.ctx):
|
||||
raise HTTPException(404, "Modell nicht gefunden")
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
@router.get("/models/drafts")
|
||||
def list_drafts(target: str = "") -> dict:
|
||||
"""Verfügbare Draft-Modelle + ihre Vocab-Kompatibilität zum Ziel-Modell
|
||||
(target = GGUF-Pfad). Basis für die idiotensichere Spec-Draft-Auswahl im UI."""
|
||||
return llamaswap.drafts_for(target)
|
||||
|
||||
|
||||
class DraftReq(BaseModel):
|
||||
draft_path: str | None = None
|
||||
|
||||
|
||||
@router.post("/models/{model_id}/draft")
|
||||
def set_model_draft(model_id: str, body: DraftReq) -> dict:
|
||||
"""Setzt/entfernt den Speculative-Decoding-Draft eines Modells. Inkompatible
|
||||
(oder nicht prüfbare) Drafts werden serverseitig abgelehnt."""
|
||||
try:
|
||||
res = llamaswap.set_spec_draft(model_id, body.draft_path)
|
||||
except PermissionError as exc:
|
||||
raise HTTPException(500, str(exc))
|
||||
if not res["ok"]:
|
||||
raise HTTPException(400 if "kompatib" in res["reason"].lower() else 404, res["reason"])
|
||||
return res
|
||||
|
||||
|
||||
@router.post("/models/unload")
|
||||
def unload_all_models() -> dict:
|
||||
import httpx
|
||||
from config import LLAMA_SWAP_URL
|
||||
try:
|
||||
with httpx.Client(timeout=10.0) as c:
|
||||
r = c.post(f"{LLAMA_SWAP_URL}/api/models/unload")
|
||||
return {"ok": r.status_code == 200}
|
||||
except Exception as exc:
|
||||
raise HTTPException(500, str(exc))
|
||||
|
||||
|
||||
@router.post("/models/{model_id}/unload")
|
||||
def unload_model(model_id: str) -> dict:
|
||||
import httpx
|
||||
from config import LLAMA_SWAP_URL
|
||||
try:
|
||||
with httpx.Client(timeout=10.0) as c:
|
||||
r = c.post(f"{LLAMA_SWAP_URL}/api/models/unload/{model_id}")
|
||||
return {"ok": r.status_code == 200}
|
||||
except Exception as exc:
|
||||
raise HTTPException(500, str(exc))
|
||||
|
||||
|
||||
@router.post("/models/{model_id}/load")
|
||||
def load_model(model_id: str) -> dict:
|
||||
import httpx
|
||||
from config import LLAMA_SWAP_URL
|
||||
try:
|
||||
# Trigger load by sending a lightweight completion request.
|
||||
body = {
|
||||
"model": model_id,
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
"max_tokens": 1
|
||||
}
|
||||
# High timeout because model loading might take time
|
||||
with httpx.Client(timeout=60.0) as c:
|
||||
c.post(f"{LLAMA_SWAP_URL}/v1/chat/completions", json=body)
|
||||
return {"ok": True}
|
||||
except Exception as exc:
|
||||
raise HTTPException(500, str(exc))
|
||||
|
||||
|
||||
@router.delete("/models/{model_id}")
|
||||
def delete(model_id: str) -> dict:
|
||||
if not llamaswap.delete_model(model_id):
|
||||
raise HTTPException(404, "Modell nicht gefunden")
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
@router.get("/groups")
|
||||
def groups() -> dict:
|
||||
return {"groups": llamaswap.list_groups()}
|
||||
|
||||
|
||||
class GroupReq(BaseModel):
|
||||
group: str
|
||||
members: list[str]
|
||||
swap: bool = False
|
||||
persist: bool = False
|
||||
|
||||
|
||||
@router.put("/groups")
|
||||
def set_group(req: GroupReq) -> dict:
|
||||
try:
|
||||
llamaswap.set_group(req.group, req.members, swap=req.swap, persist=req.persist)
|
||||
except PermissionError as exc:
|
||||
raise HTTPException(500, str(exc))
|
||||
return {"ok": True}
|
||||
@@ -0,0 +1,43 @@
|
||||
"""Routing-Endpoints: Lane-Summary (chat/coding) + UI-editierbare Policy (hot-reload)."""
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
from services import gateway
|
||||
from services.routing_policy import policy_meta, save_policy
|
||||
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
|
||||
@router.get("/routing")
|
||||
def routing() -> dict:
|
||||
return {**gateway.routing_summary(), "gateway_reachable": gateway.gateway_reachable()}
|
||||
|
||||
|
||||
@router.get("/routing/policy")
|
||||
def get_policy() -> dict:
|
||||
"""Aktuelle Policy + Defaults (für „Zurücksetzen“) + Feld-Spezifikation für den Editor."""
|
||||
return policy_meta()
|
||||
|
||||
|
||||
class PolicyPatch(BaseModel):
|
||||
fast: str | None = None
|
||||
heavy: str | None = None
|
||||
coder: str | None = None
|
||||
coder_lite: str | None = None
|
||||
heavy_chars: int | None = None
|
||||
coding_escalate_chars: int | None = None
|
||||
fast_no_think: bool | None = None
|
||||
|
||||
|
||||
@router.put("/routing/policy")
|
||||
def put_policy(patch: PolicyPatch) -> dict:
|
||||
"""Teil-Update der Routing-Policy. Validiert, persistiert atomar, sofort wirksam (hot-reload)."""
|
||||
fields = {k: v for k, v in patch.model_dump().items() if v is not None}
|
||||
if not fields:
|
||||
raise HTTPException(status_code=400, detail="Keine Felder zum Aktualisieren.")
|
||||
try:
|
||||
new_policy = save_policy(fields)
|
||||
except (ValueError, TypeError) as e:
|
||||
raise HTTPException(status_code=400, detail=f"Ungültige Policy: {e}")
|
||||
return {"policy": new_policy}
|
||||
@@ -0,0 +1,128 @@
|
||||
"""System-Endpoints: Live-Status + Wartung (Restart/Self-Update — auf der Box).
|
||||
|
||||
Wartung läuft als systemd-USER-Dienst → KEIN sudo/Passwort (Nordstern).
|
||||
Lokal (Windows) schlagen die Shell-Befehle harmlos fehl und werden als Fehler
|
||||
zurückgegeben statt zu crashen.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
from fastapi import APIRouter, HTTPException
|
||||
from pydantic import BaseModel
|
||||
|
||||
import httpx
|
||||
|
||||
from config import GATEWAY_URL, HERMES_API_URL, LLAMA_SWAP_URL, MEM0_SERVICE_URL, VOICE_SERVICE_URL
|
||||
from services import backup as backup_svc
|
||||
from services.agent import agent_status
|
||||
from services.gateway import gateway_reachable
|
||||
from services.llamaswap import engine_reachable, list_models
|
||||
from services.pricing import compute_savings
|
||||
from services.system import system_status
|
||||
from services.token_stats import get_stats
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
# Nur diese User-Dienste dürfen neugestartet werden.
|
||||
ALLOWED_SERVICES = {"mission-control-2", "hermes-gateway", "hermes-webui", "mem0-service", "voice-service"}
|
||||
# Quelle für Self-Update (auf der Box ~/mission-control-v2).
|
||||
SOURCE_DIR = os.path.expanduser(os.environ.get("MC2_SOURCE_DIR", "~/mission-control-v2"))
|
||||
|
||||
|
||||
@router.get("/system/status")
|
||||
def status() -> dict:
|
||||
return system_status()
|
||||
|
||||
|
||||
def _mem0_reachable() -> bool:
|
||||
try:
|
||||
return httpx.get(f"{MEM0_SERVICE_URL}/health", timeout=2).status_code == 200
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _voice_reachable() -> bool:
|
||||
try:
|
||||
return httpx.get(f"{VOICE_SERVICE_URL}/health", timeout=2).status_code == 200
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
@router.get("/system/services")
|
||||
def services() -> dict:
|
||||
"""Aggregierte Erreichbarkeit aller Stack-Dienste (für die Health-Anzeige)."""
|
||||
a = agent_status()
|
||||
gw_url = f"{GATEWAY_URL}/v1"
|
||||
return {
|
||||
"services": [
|
||||
{"name": "Engine (llama-swap)", "url": LLAMA_SWAP_URL, "ok": engine_reachable()},
|
||||
{"name": "Gateway (integriert)", "url": gw_url, "ok": gateway_reachable()},
|
||||
{"name": "Hermes-Gateway", "url": HERMES_API_URL, "ok": a["gateway_reachable"]},
|
||||
{"name": "Hermes-Terminal", "url": a["terminal_url"], "ok": a["terminal_reachable"]},
|
||||
{"name": "Mem0 (Gedächtnis)", "url": MEM0_SERVICE_URL, "ok": _mem0_reachable()},
|
||||
{"name": "Voice (STT/TTS)", "url": VOICE_SERVICE_URL, "ok": _voice_reachable()},
|
||||
],
|
||||
"links": {
|
||||
"engine_ui": f"{LLAMA_SWAP_URL}/ui",
|
||||
"gateway": gw_url,
|
||||
"hermes_terminal": a["terminal_url"],
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@router.post("/system/backup")
|
||||
def backup() -> dict:
|
||||
return backup_svc.backup_now()
|
||||
|
||||
|
||||
@router.get("/system/backups")
|
||||
def backups() -> dict:
|
||||
return {"backups": backup_svc.list_backups()}
|
||||
|
||||
|
||||
def _run(cmd: list[str], cwd: str | None = None) -> dict:
|
||||
try:
|
||||
p = subprocess.run(cmd, cwd=cwd, capture_output=True, text=True, timeout=180)
|
||||
return {"ok": p.returncode == 0, "code": p.returncode,
|
||||
"out": (p.stdout or "")[-2000:], "err": (p.stderr or "")[-2000:]}
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {"ok": False, "code": -1, "out": "", "err": str(exc)}
|
||||
|
||||
|
||||
class RestartReq(BaseModel):
|
||||
service: str
|
||||
|
||||
|
||||
@router.post("/system/restart")
|
||||
def restart(req: RestartReq) -> dict:
|
||||
if req.service not in ALLOWED_SERVICES:
|
||||
raise HTTPException(400, f"Dienst '{req.service}' nicht erlaubt.")
|
||||
return _run(["systemctl", "--user", "restart", req.service])
|
||||
|
||||
|
||||
@router.post("/system/self-update")
|
||||
def self_update() -> dict:
|
||||
"""git pull (Source) → venv-Deps → Dienst-Restart. Auf der Box; lokal Fehler."""
|
||||
pull = _run(["git", "fetch", "--all"], cwd=SOURCE_DIR)
|
||||
reset = _run(["git", "reset", "--hard", "origin/main"], cwd=SOURCE_DIR)
|
||||
restart_res = _run(["systemctl", "--user", "restart", "mission-control-2"])
|
||||
return {"pull": pull, "reset": reset, "restart": restart_res}
|
||||
|
||||
|
||||
@router.get("/system/token-stats")
|
||||
def token_stats() -> dict:
|
||||
"""Token-Verbrauch + Cloud-Ersparnis. Logik im pricing-Service (SSoT)."""
|
||||
# Rolle je Modell/Alias (lowercase) für die Tarif-Auflösung auflösen.
|
||||
role_map: dict[str, str | None] = {}
|
||||
try:
|
||||
for m in list_models():
|
||||
role_map[m["name"].lower()] = m.get("role")
|
||||
for alias in m.get("aliases", []):
|
||||
role_map[alias.lower()] = m.get("role")
|
||||
except Exception:
|
||||
log.warning("token_stats: list_models fehlgeschlagen, Tarife per Name", exc_info=True)
|
||||
return compute_savings(get_stats(), role_map)
|
||||
@@ -0,0 +1,272 @@
|
||||
"""
|
||||
Voice-Endpoints für „Mit Hermes reden" (Browser-Voice + 3D-Avatar).
|
||||
|
||||
Dünner Layer: STT/TTS werden zum Voice-Sidecar (:8650) geproxyt; der Chat geht an den
|
||||
Hermes-`api_server` (:8642, OpenAI-kompatibel) — denselben vollen Agenten mit Tools +
|
||||
geteiltem Mem0 wie CLI/Telegram. Mit stabilem `X-Hermes-Session-Id` hält die Plattform den
|
||||
Transcript server-seitig, daher schickt der Client je Turn nur die neue User-Nachricht.
|
||||
|
||||
LAN-only (kein Token in der 2.0-Phase), wie die übrigen MC2-Endpoints.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
|
||||
import httpx
|
||||
from fastapi import APIRouter, File, Form, HTTPException, UploadFile
|
||||
from fastapi.responses import Response, StreamingResponse
|
||||
from pydantic import BaseModel
|
||||
|
||||
from config import HERMES_API_KEY, HERMES_API_MODEL, HERMES_API_URL, LLAMA_SWAP_URL, VOICE_SERVICE_URL
|
||||
from services.voice_metrics import Timer, get_metrics, record_stage # Per-Stage-Latenz (C2)
|
||||
|
||||
# Injection-Schutz (Stufe 0): guard.py liegt im mcp/-Verzeichnis. Per Pfad laden (eigene MC2-Venv).
|
||||
import sys as _sys
|
||||
_GUARD_DIR = os.path.join(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))), "mcp")
|
||||
if _GUARD_DIR not in _sys.path:
|
||||
_sys.path.insert(0, _GUARD_DIR)
|
||||
try:
|
||||
from guard import wrap_untrusted
|
||||
except Exception: # den Voice-Pfad nie wegen des Filters lahmlegen
|
||||
def wrap_untrusted(text: str, label: str = "") -> str:
|
||||
return text
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/api")
|
||||
|
||||
# Bildschirm-Sicht: das DEDIZIERTE Vision-Modell (Qwen3-VL-8B) beschreibt das Bild; die Beschreibung
|
||||
# geht als TEXT an Hermes -> Lucy behält ihr volles Hirn/Gedächtnis UND nutzt das bessere VL-Modell
|
||||
# (statt der schwächeren Vision der fast-MoE). Per Env abschaltbar/umstellbar.
|
||||
VISION_MODEL = os.environ.get("MC_VISION_MODEL", "vision")
|
||||
# Knappe Beschreibung = schnellere VL-Generierung UND weniger Hermes-Kontext-Bloat (B2).
|
||||
VISION_MAX_TOKENS = int(os.environ.get("MC_VISION_MAX_TOKENS", "280"))
|
||||
|
||||
|
||||
async def _describe_images(image_urls: list[str], hint: str) -> str:
|
||||
"""Lässt das Vision-Modell die Screenshots (1 je Monitor) knapp beschreiben (Deutsch).
|
||||
Mehrere Bilder gehen in EINER Nachricht ans VL-Modell. Leerer String bei Fehler."""
|
||||
multi = len(image_urls) > 1
|
||||
intro = (f"Hier sind {len(image_urls)} Screenshots (je ein Monitor). Beschreibe auf Deutsch in höchstens "
|
||||
"5 kurzen Sätzen das Wesentliche (pro Monitor: App/Fenster, wichtige Inhalte, sichtbarer Text/Code). "
|
||||
"Keine Einleitung, keine Wiederholung der Frage. "
|
||||
if multi else
|
||||
"Beschreibe auf Deutsch in höchstens 5 kurzen Sätzen das Wesentliche auf diesem Screenshot "
|
||||
"(App/Fenster, wichtige Inhalte, sichtbarer Text/Code). Keine Einleitung. ")
|
||||
content: list = [{"type": "text", "text": intro + "Frage des Nutzers dazu: " + hint}]
|
||||
for u in image_urls:
|
||||
content.append({"type": "image_url", "image_url": {"url": u}})
|
||||
try:
|
||||
# 45 s statt 120 s: Qwen3-VL braucht warm ~5 s; wenn es 45 s nicht schafft, ist etwas
|
||||
# kaputt und Lucy soll lieber ohne Bildschirm-Kontext antworten als ewig hängen.
|
||||
async with httpx.AsyncClient(timeout=httpx.Timeout(float(os.environ.get("MC_VISION_TIMEOUT", "45")), connect=5.0)) as client:
|
||||
r = await client.post(f"{LLAMA_SWAP_URL}/v1/chat/completions", json={
|
||||
"model": VISION_MODEL, "max_tokens": VISION_MAX_TOKENS, "stream": False,
|
||||
"messages": [{"role": "user", "content": content}],
|
||||
})
|
||||
r.raise_for_status()
|
||||
return (r.json().get("choices") or [{}])[0].get("message", {}).get("content", "").strip()
|
||||
except Exception as exc:
|
||||
log.warning("Vision-Beschreibung fehlgeschlagen: %s", exc)
|
||||
return ""
|
||||
|
||||
_TIMEOUT = httpx.Timeout(120.0, connect=5.0) # Chatterbox-TTS auf CPU darf dauern
|
||||
|
||||
|
||||
class TTSIn(BaseModel):
|
||||
text: str
|
||||
engine: str = "piper"
|
||||
voice: str = ""
|
||||
language: str = ""
|
||||
ref_path: str = ""
|
||||
|
||||
|
||||
class ChatIn(BaseModel):
|
||||
text: str # die neue User-Äußerung (STT-Ergebnis)
|
||||
session_id: str # stabiler Voice-Faden → server-seitiger Transcript
|
||||
session_key: str = "" # optional: Langzeit-Memory-Scope
|
||||
system: str = "" # optionaler ephemerer System-Prompt (z.B. „antworte knapp/gesprochen")
|
||||
model: str = ""
|
||||
images: list[str] = [] # optionale Bildschirm-Sicht: ein data:-URL je Monitor (Lucys „Augen")
|
||||
|
||||
|
||||
@router.get("/voice/metrics")
|
||||
def voice_metrics() -> dict:
|
||||
"""Per-Stage-Latenz (STT/Vision/Chat-TTFB/TTS) — rollende Statistik, macht die Voice-Pipeline
|
||||
messbar (C2). Anzeige im Frontend-Overhaul (E)."""
|
||||
return get_metrics()
|
||||
|
||||
|
||||
@router.get("/voice/health")
|
||||
def voice_health() -> dict:
|
||||
"""Erreichbarkeit des Voice-Sidecars + ob der Hermes-API-Key gesetzt ist."""
|
||||
out: dict = {"sidecar": False, "hermes_key": bool(HERMES_API_KEY)}
|
||||
try:
|
||||
r = httpx.get(f"{VOICE_SERVICE_URL}/health", timeout=httpx.Timeout(5.0))
|
||||
out["sidecar"] = r.status_code == 200
|
||||
out["detail"] = r.json() if r.status_code == 200 else None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
out["error"] = str(exc)
|
||||
return out
|
||||
|
||||
|
||||
@router.get("/voice/voices")
|
||||
def voice_voices() -> dict:
|
||||
try:
|
||||
r = httpx.get(f"{VOICE_SERVICE_URL}/voices", timeout=httpx.Timeout(10.0))
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise HTTPException(502, f"Voice-Sidecar nicht erreichbar: {exc}")
|
||||
|
||||
|
||||
@router.post("/voice/stt")
|
||||
async def voice_stt(audio: UploadFile = File(...), language: str = Form(default="")) -> dict:
|
||||
"""Mikro-Audio → Text (Proxy auf Sidecar /stt)."""
|
||||
data = await audio.read()
|
||||
if not data:
|
||||
raise HTTPException(400, "Leeres Audio.")
|
||||
files = {"audio": (audio.filename or "rec.webm", data, audio.content_type or "audio/webm")}
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=_TIMEOUT) as client:
|
||||
with Timer("stt"):
|
||||
r = await client.post(f"{VOICE_SERVICE_URL}/stt", files=files, data={"language": language})
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
except httpx.HTTPError as exc:
|
||||
raise HTTPException(502, f"STT fehlgeschlagen: {exc}")
|
||||
|
||||
|
||||
@router.post("/voice/turn")
|
||||
async def voice_turn(audio: UploadFile = File(...)) -> dict:
|
||||
"""Semantische Turn-Detection (Smart Turn v3): war die Äußerung fertig? Proxy → Sidecar."""
|
||||
data = await audio.read()
|
||||
if not data:
|
||||
raise HTTPException(400, "Leeres Audio.")
|
||||
files = {"audio": (audio.filename or "rec.wav", data, audio.content_type or "audio/wav")}
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=httpx.Timeout(10.0, connect=3.0)) as client:
|
||||
with Timer("turn"):
|
||||
r = await client.post(f"{VOICE_SERVICE_URL}/turn", files=files)
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
except httpx.HTTPError as exc:
|
||||
# Turn-Check ist eine Optimierung — bei Ausfall lieber sofort antworten als hängen.
|
||||
log.warning("Turn-Check fehlgeschlagen: %s", exc)
|
||||
return {"complete": True, "probability": 1.0, "engine": "fallback"}
|
||||
|
||||
|
||||
@router.post("/voice/reference")
|
||||
async def voice_set_reference(audio: UploadFile = File(...)) -> dict:
|
||||
"""Klon-Referenz (z.B. ElevenLabs-Erzeugnis) hochladen → Chatterbox nutzt sie. Proxy → Sidecar."""
|
||||
data = await audio.read()
|
||||
if not data:
|
||||
raise HTTPException(400, "Leeres Audio.")
|
||||
files = {"audio": (audio.filename or "ref.wav", data, audio.content_type or "audio/mpeg")}
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=_TIMEOUT) as client:
|
||||
r = await client.post(f"{VOICE_SERVICE_URL}/reference", files=files)
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
except httpx.HTTPError as exc:
|
||||
raise HTTPException(502, f"Referenz-Upload fehlgeschlagen: {exc}")
|
||||
|
||||
|
||||
@router.get("/voice/reference")
|
||||
def voice_get_reference() -> dict:
|
||||
try:
|
||||
r = httpx.get(f"{VOICE_SERVICE_URL}/reference", timeout=httpx.Timeout(8.0))
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {"active": False, "error": str(exc)}
|
||||
|
||||
|
||||
@router.delete("/voice/reference")
|
||||
def voice_clear_reference() -> dict:
|
||||
try:
|
||||
r = httpx.delete(f"{VOICE_SERVICE_URL}/reference", timeout=httpx.Timeout(8.0))
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
except httpx.HTTPError as exc:
|
||||
raise HTTPException(502, f"Löschen fehlgeschlagen: {exc}")
|
||||
|
||||
|
||||
@router.post("/voice/tts")
|
||||
async def voice_tts(body: TTSIn) -> Response:
|
||||
"""Text → Sprache (Proxy auf Sidecar /tts), liefert WAV-Bytes."""
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=_TIMEOUT) as client:
|
||||
with Timer("tts"):
|
||||
r = await client.post(f"{VOICE_SERVICE_URL}/tts", json=body.model_dump())
|
||||
r.raise_for_status()
|
||||
return Response(content=r.content, media_type=r.headers.get("content-type", "audio/wav"))
|
||||
except httpx.HTTPError as exc:
|
||||
raise HTTPException(502, f"TTS fehlgeschlagen: {exc}")
|
||||
|
||||
|
||||
@router.post("/voice/chat")
|
||||
async def voice_chat(body: ChatIn) -> StreamingResponse:
|
||||
"""Neue User-Äußerung → Hermes-Agent (api_server, streamend). SSE wird 1:1 durchgereicht.
|
||||
|
||||
Mit `X-Hermes-Session-Id` hält die Plattform den Verlauf — wir senden nur die neue Nachricht.
|
||||
Auth per Bearer (API_SERVER_KEY); ohne Key liefert :8642 ein 401."""
|
||||
if not HERMES_API_KEY:
|
||||
raise HTTPException(503, "HERMES_API_KEY/API_SERVER_KEY nicht gesetzt — Agent-Auth fehlt.")
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {HERMES_API_KEY}",
|
||||
"X-Hermes-Session-Id": body.session_id,
|
||||
}
|
||||
if body.session_key:
|
||||
headers["X-Hermes-Session-Key"] = body.session_key
|
||||
|
||||
async def gen():
|
||||
t0 = time.perf_counter()
|
||||
first = True
|
||||
first_content = True
|
||||
# Bildschirm-Sicht INNERHALB des Streams (C2-Fix): so startet die SSE-Antwort sofort und
|
||||
# der Client bekommt ein Progress-Event (-> Lucy kann eine Warte-Ansage sprechen), statt
|
||||
# dass der Request bis zu 120 s "tot" hängt, während das Vision-Modell beschreibt.
|
||||
user_text = body.text
|
||||
imgs = [u for u in (body.images or []) if u]
|
||||
if imgs:
|
||||
yield b'event: hermes.vision.progress\ndata: {"note": "Bildschirm wird angeschaut"}\n\n'
|
||||
with Timer("vision"):
|
||||
desc = await _describe_images(imgs, body.text)
|
||||
if desc:
|
||||
safe_desc = wrap_untrusted(desc, "BILDSCHIRM")
|
||||
user_text = f"[Bildschirm-Sicht — das ist gerade auf dem/den Schirm(en) zu sehen:\n{safe_desc}\n]\n\n{body.text}"
|
||||
messages = []
|
||||
if body.system:
|
||||
messages.append({"role": "system", "content": body.system})
|
||||
messages.append({"role": "user", "content": user_text})
|
||||
payload = {"model": body.model or HERMES_API_MODEL, "messages": messages, "stream": True}
|
||||
# Lucys Hirn (Qwen3.6) ist ein Thinking-Modell -> für die gesprochene Assistentin Thinking AUS,
|
||||
# sonst generiert es tausende Reasoning-Token VOR der kurzen Antwort (gemessen: 11k Token, ~30s TTFB).
|
||||
# Gleiches Muster wie die fast-Spur im Gateway (gateway_proxy.py) und die Mem0-Extraktion.
|
||||
if os.environ.get("MC_VOICE_NO_THINK", "1") not in ("0", "false", "False"):
|
||||
payload["chat_template_kwargs"] = {"enable_thinking": False}
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=httpx.Timeout(None, connect=5.0)) as client:
|
||||
async with client.stream(
|
||||
"POST", f"{HERMES_API_URL}/v1/chat/completions", json=payload, headers=headers,
|
||||
) as r:
|
||||
if r.status_code != 200:
|
||||
detail = (await r.aread()).decode("utf-8", "replace")[:500]
|
||||
yield f"data: {{\"error\": \"Hermes {r.status_code}: {detail}\"}}\n\n".encode()
|
||||
return
|
||||
async for chunk in r.aiter_raw():
|
||||
if first: # Time-To-First-Byte des Hermes-Streams (Verbindungs-Overhead)
|
||||
record_stage("chat_ttfb", (time.perf_counter() - t0) * 1000.0)
|
||||
first = False
|
||||
# Erster CONTENT-Delta = echte Hirn-Latenz (Agent-Overhead + LLM-TTFT) —
|
||||
# chat_ttfb misst nur den SSE-Start (~5 ms) und ist dafür blind.
|
||||
if first_content and b'"content"' in chunk:
|
||||
record_stage("chat_first_content", (time.perf_counter() - t0) * 1000.0)
|
||||
first_content = False
|
||||
yield chunk
|
||||
except httpx.HTTPError as exc:
|
||||
yield f"data: {{\"error\": \"Verbindung zu Hermes fehlgeschlagen: {exc}\"}}\n\n".encode()
|
||||
|
||||
return StreamingResponse(gen(), media_type="text/event-stream")
|
||||
@@ -0,0 +1,252 @@
|
||||
"""
|
||||
Hermes-Agent-Status (Control-Plane-Read). MC betreibt Hermes NICHT — es zeigt nur
|
||||
Status + verlinkt das standalone hermes-webui. Voller Zugriff + Tools/MCP werden in
|
||||
Hermes' eigener Config verdrahtet (siehe docs/HERMES_SETUP.md).
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
|
||||
import httpx
|
||||
import psutil
|
||||
|
||||
from config import HERMES_TERMINAL_URL, HERMES_API_URL, HERMES_HOME, PC_EXECUTOR_URL
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _hermes_version(name: str) -> float | None:
|
||||
"""Versionszahl aus 'Hermes-4.3', 'Hermes-4', 'Nous-Hermes-2' → 4.3/4.0/2.0."""
|
||||
low = (name or "").lower()
|
||||
if "hermes" not in low:
|
||||
return None
|
||||
m = re.search(r"hermes[-_ ]?(\d+(?:\.\d+)?)", low)
|
||||
return float(m.group(1)) if m else None
|
||||
|
||||
|
||||
def _active_brain_name() -> str:
|
||||
"""Aktives Agent-Hirn aus Hermes' Config: model.default (sonst model.model)."""
|
||||
try:
|
||||
from ruamel.yaml import YAML
|
||||
p = HERMES_HOME / "config.yaml"
|
||||
if p.exists():
|
||||
with p.open(encoding="utf-8") as f:
|
||||
cfg = YAML().load(f) or {}
|
||||
m = (cfg.get("model") or {}) if isinstance(cfg, dict) else {}
|
||||
return str(m.get("default") or m.get("model") or "auto")
|
||||
except Exception:
|
||||
log.debug("_active_brain_name: Lesefehler", exc_info=True)
|
||||
return "auto"
|
||||
|
||||
|
||||
def hermes_brain_info() -> dict:
|
||||
"""Aktuelles Agent-Hirn = Modell/Alias, das Hermes laut Config nutzt (model.default),
|
||||
plus Budget-Check. Zeigt das REAL genutzte Hirn — unabhängig von einer 'hermes'-Rolle."""
|
||||
from services import llamaswap
|
||||
|
||||
models = llamaswap.list_models()
|
||||
brain = _active_brain_name() # z.B. "fast" (Alias) oder ein Modellname
|
||||
bl = brain.lower()
|
||||
cur = next((m for m in models if (m.get("role") or "").lower() == bl), None) \
|
||||
or next((m for m in models if bl in (m["name"] or "").lower()), None)
|
||||
cur_params = (cur.get("capabilities") or {}).get("params_b") if cur else None
|
||||
current = None
|
||||
if cur:
|
||||
current = {"name": cur["name"], "alias": brain, "filename": cur.get("filename"),
|
||||
"params_b": cur_params, "quant": cur.get("quant"),
|
||||
"size_bytes": cur.get("size_bytes"),
|
||||
"gguf_path": cur.get("gguf_path"), "incomplete": cur.get("incomplete")}
|
||||
|
||||
# Fit-Check: passt das (immer warme) Hirn + das größte on-demand-Modell zusammen ins Budget?
|
||||
budget = None
|
||||
try:
|
||||
from services.budget import footprint_gb, gtt_budget_gb
|
||||
groups = llamaswap.list_groups()
|
||||
persist = set()
|
||||
for g in groups.values():
|
||||
if isinstance(g, dict) and g.get("persist"):
|
||||
persist.update(g.get("members") or [])
|
||||
|
||||
cur_name = cur["name"] if cur else None
|
||||
brain_gb = footprint_gb(cur) if cur else 0.0
|
||||
# voller Always-Warm-Footprint (alle persist, Brain=Empfehlung) — nur Info
|
||||
warm = brain_gb + sum(footprint_gb(m) for m in models
|
||||
if m["name"] in persist and m["name"] != cur_name)
|
||||
largest_od = max((footprint_gb(m) for m in models if m["name"] not in persist), default=0.0)
|
||||
gtt = gtt_budget_gb()
|
||||
# Brain muss immer resident sein → passt Brain + größtes on-demand zusammen?
|
||||
# (fast/vision dürfen beim Laden eines großen Modells verdrängt werden.)
|
||||
budget = {
|
||||
"gtt_gb": gtt,
|
||||
"brain_gb": round(brain_gb, 1),
|
||||
"warm_projected_gb": round(warm, 1),
|
||||
"largest_ondemand_gb": round(largest_od, 1),
|
||||
"fits": (brain_gb + largest_od) <= gtt,
|
||||
"free_after_gb": round(gtt - brain_gb - largest_od, 1),
|
||||
}
|
||||
except Exception:
|
||||
log.debug("hermes_brain_info: Budget-Berechnung fehlgeschlagen", exc_info=True)
|
||||
|
||||
return {"current": current, "recommended": None, "update_available": False, "budget": budget}
|
||||
|
||||
|
||||
def _reach(url: str, path: str = "") -> bool:
|
||||
try:
|
||||
with httpx.Client(timeout=3.0) as c:
|
||||
return c.get(f"{url}{path}").status_code < 500
|
||||
except httpx.HTTPError:
|
||||
return False
|
||||
|
||||
|
||||
def _count_enabled_mcp_servers() -> int:
|
||||
config_path = HERMES_HOME / "config.yaml"
|
||||
if not config_path.exists():
|
||||
return 0
|
||||
try:
|
||||
from ruamel.yaml import YAML
|
||||
r_yaml = YAML()
|
||||
with config_path.open("r", encoding="utf-8") as f:
|
||||
cfg = r_yaml.load(f) or {}
|
||||
mcp_servers = cfg.get("mcp_servers", {}) if isinstance(cfg, dict) else {}
|
||||
if not isinstance(mcp_servers, dict):
|
||||
return 0
|
||||
return sum(1 for v in mcp_servers.values() if isinstance(v, dict) and v.get("enabled", True))
|
||||
except Exception:
|
||||
log.debug("_count_enabled_mcp_servers: Fehler", exc_info=True)
|
||||
return 0
|
||||
|
||||
|
||||
def agent_status() -> dict:
|
||||
"""Erreichbarkeit von Gateway (:8642) + WebUI (:8787) + lokale Hinweise."""
|
||||
home = HERMES_HOME
|
||||
brain_model = "auto"
|
||||
config_path = home / "config.yaml"
|
||||
if config_path.exists():
|
||||
try:
|
||||
from ruamel.yaml import YAML
|
||||
r_yaml = YAML()
|
||||
with config_path.open("r", encoding="utf-8") as f:
|
||||
cfg = r_yaml.load(f) or {}
|
||||
if isinstance(cfg, dict):
|
||||
# Hermes nutzt model.default als aktives Modell (model.model = Provider-Param).
|
||||
m = cfg.get("model", {}) or {}
|
||||
brain_model = m.get("default") or m.get("model") or "auto"
|
||||
except Exception:
|
||||
log.debug("agent_status: Hermes-config.yaml nicht lesbar", exc_info=True)
|
||||
|
||||
|
||||
return {
|
||||
"gateway_url": HERMES_API_URL,
|
||||
# Interaktives Web-Terminal (ttyd → `hermes chat`), eingebettet in MC2.
|
||||
"terminal_url": HERMES_TERMINAL_URL,
|
||||
"gateway_reachable": _reach(HERMES_API_URL, "/health"),
|
||||
"terminal_reachable": _reach(HERMES_TERMINAL_URL, "/"),
|
||||
"home_exists": home.exists(),
|
||||
"brain_model": brain_model,
|
||||
# Best-effort: welche Verdrahtung lokal sichtbar ist (auf der Box aussagekräftig).
|
||||
"has_config": (home / "config.yaml").exists() or (home / "config.json").exists(),
|
||||
"has_skills": (home / "skills").exists(),
|
||||
"has_memories": (home / "memories").exists(),
|
||||
# Neue Felder: Telegram, MCP-Server-Anzahl, PC-Executor-Erreichbarkeit.
|
||||
"telegram_enabled": bool(os.environ.get("TELEGRAM_BOT_TOKEN", "")),
|
||||
"mcp_server_count": _count_enabled_mcp_servers(),
|
||||
"pc_executor_reachable": _reach(PC_EXECUTOR_URL, "/health"),
|
||||
}
|
||||
|
||||
|
||||
def set_agent_brain(model_id: str) -> dict:
|
||||
"""Setzt ein (bereits installiertes) Modell als Agent-Hirn — WARM-bewusst:
|
||||
1) vergibt den 'hermes'-Alias (das Agent-Hirn-Slot),
|
||||
2) tauscht es in die residente brains-Gruppe (altes Hirn raus, fast/vision bleiben),
|
||||
3) zeigt die Hermes-Config auf den 'hermes'-Alias + Gateway-Restart.
|
||||
So bleibt das neue Hirn warm und der Agent nutzt es sofort."""
|
||||
from services import llamaswap
|
||||
models = {m["name"]: m for m in llamaswap.list_models()}
|
||||
if model_id not in models:
|
||||
return {"ok": False, "reason": "Modell nicht installiert — erst über Modelle-finden laden."}
|
||||
old = next((m["name"] for m in models.values() if m.get("role") == "hermes"), None)
|
||||
if model_id == old:
|
||||
# Idempotent härten: auch wenn schon Hirn, warm (brains) + ttl 0 sicherstellen.
|
||||
try:
|
||||
from services.llamaswap import set_ttl
|
||||
brains = (llamaswap.list_groups().get("brains") or {}).get("members") or []
|
||||
if model_id not in brains:
|
||||
llamaswap.set_group("brains", brains + [model_id], swap=False, persist=True)
|
||||
set_ttl(model_id, 0)
|
||||
except PermissionError as exc:
|
||||
return {"ok": False, "reason": str(exc)}
|
||||
return {"ok": True, "old": old, "new": model_id, "note": "ist bereits das Agent-Hirn"}
|
||||
try:
|
||||
llamaswap.set_role(model_id, "hermes") # 1) Alias
|
||||
brains = (llamaswap.list_groups().get("brains") or {}).get("members") or []
|
||||
new_members = [x for x in brains if x not in (old, model_id)] + [model_id]
|
||||
llamaswap.set_group("brains", new_members, swap=False, persist=True) # 2) warm
|
||||
# 2b) TTL härten: neues Hirn nie auto-entladen; altes Hirn auf Default entspannen.
|
||||
from services.llamaswap import set_ttl, DEFAULT_TTL
|
||||
set_ttl(model_id, 0)
|
||||
if old:
|
||||
set_ttl(old, DEFAULT_TTL)
|
||||
except PermissionError as exc:
|
||||
return {"ok": False, "reason": str(exc)}
|
||||
update_brain_model("hermes") # 3) Config + Restart
|
||||
# Weiche Budget-Warnung (kein Hard-Block): passt Hirn + größtes on-demand zusammen ins GTT?
|
||||
warning = None
|
||||
try:
|
||||
b = hermes_brain_info().get("budget") or {}
|
||||
if b and not b.get("fits", True):
|
||||
warning = (f"Speicher-Warnung: Hirn (~{b.get('brain_gb')} GB) + größtes on-demand-"
|
||||
f"Modell (~{b.get('largest_ondemand_gb')} GB) übersteigen das GTT-Budget "
|
||||
f"(~{b.get('gtt_gb')} GB) — heavy/coder würden das Hirn verdrängen.")
|
||||
except Exception:
|
||||
log.debug("set_agent_brain: Budget-Check fehlgeschlagen", exc_info=True)
|
||||
return {"ok": True, "old": old, "new": model_id, "warning": warning}
|
||||
|
||||
|
||||
def update_brain_model(new_model: str) -> bool:
|
||||
from config import HERMES_HOME
|
||||
home = HERMES_HOME
|
||||
config_path = home / "config.yaml"
|
||||
|
||||
# Ensure home directory exists
|
||||
home.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
cfg = {}
|
||||
if config_path.exists():
|
||||
try:
|
||||
from ruamel.yaml import YAML
|
||||
r_yaml = YAML()
|
||||
with config_path.open("r", encoding="utf-8") as f:
|
||||
cfg = r_yaml.load(f) or {}
|
||||
except Exception:
|
||||
log.debug("update_brain_model: bestehende config.yaml nicht lesbar", exc_info=True)
|
||||
cfg = {}
|
||||
|
||||
if not isinstance(cfg, dict):
|
||||
cfg = {}
|
||||
|
||||
if "model" not in cfg or not isinstance(cfg["model"], dict):
|
||||
cfg["model"] = {}
|
||||
|
||||
# Hermes liest model.default als aktives Modell; model.model ist der Provider-Param.
|
||||
# Beide setzen, sonst greift die Umschaltung nicht (latenter Bug: nur model.model gesetzt).
|
||||
cfg["model"]["default"] = new_model
|
||||
cfg["model"]["model"] = new_model
|
||||
|
||||
try:
|
||||
from ruamel.yaml import YAML
|
||||
r_yaml = YAML()
|
||||
with config_path.open("w", encoding="utf-8") as f:
|
||||
r_yaml.dump(cfg, f)
|
||||
|
||||
# Restart the user-space service to apply changes
|
||||
try:
|
||||
import services.maintenance as maintenance
|
||||
maintenance.restart_service("hermes-gateway")
|
||||
except Exception:
|
||||
log.warning("update_brain_model: hermes-gateway-Restart fehlgeschlagen", exc_info=True)
|
||||
|
||||
return True
|
||||
except Exception:
|
||||
log.warning("update_brain_model: Schreiben der config.yaml fehlgeschlagen", exc_info=True)
|
||||
return False
|
||||
@@ -0,0 +1,73 @@
|
||||
"""
|
||||
Voll-Zustands-Backup (mem0 + Hermes-Configs/Secrets + llama-swap config).
|
||||
Delegiert an deploy/backup.sh (eine Quelle der Wahrheit, identisch zum systemd-Timer);
|
||||
Restore läuft bewusst nur per CLI (deploy/restore.sh) — siehe docs/BACKUP.md.
|
||||
"""
|
||||
|
||||
import subprocess
|
||||
import tarfile
|
||||
from pathlib import Path
|
||||
|
||||
from config import MODELS_DIR
|
||||
|
||||
BACKUP_DIR = Path(MODELS_DIR) / "mc2-backups"
|
||||
SRC_ROOT = Path(__file__).resolve().parents[2]
|
||||
BACKUP_SH = SRC_ROOT / "deploy" / "backup.sh"
|
||||
|
||||
|
||||
def _ts(p: Path) -> str:
|
||||
"""Zeitstempel aus 'mc2-state-<ts>.tar.gz' (Path.stem ließe '.tar' stehen)."""
|
||||
return p.name[len("mc2-state-"):-len(".tar.gz")]
|
||||
|
||||
|
||||
def _latest() -> Path | None:
|
||||
if not BACKUP_DIR.exists():
|
||||
return None
|
||||
snaps = sorted(BACKUP_DIR.glob("mc2-state-*.tar.gz"), reverse=True)
|
||||
return snaps[0] if snaps else None
|
||||
|
||||
|
||||
def _components(tarball: Path) -> list[str]:
|
||||
"""Top-Level-Einträge im Tarball (zur Anzeige im UI)."""
|
||||
try:
|
||||
with tarfile.open(tarball, "r:gz") as t:
|
||||
top = {m.name.split("/")[1] for m in t.getmembers()
|
||||
if m.name.startswith("./") and "/" in m.name[2:]}
|
||||
top |= {m.name[2:] for m in t.getmembers()
|
||||
if m.name.startswith("./") and "/" not in m.name[2:] and m.isfile()}
|
||||
return sorted(x for x in top if x)
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
def backup_now() -> dict:
|
||||
"""Erstellt einen Voll-Zustands-Snapshot via deploy/backup.sh."""
|
||||
try:
|
||||
r = subprocess.run(["/bin/bash", str(BACKUP_SH)], capture_output=True, text=True, timeout=180)
|
||||
if r.returncode != 0:
|
||||
return {"ok": False, "snapshot": "", "files": [], "error": (r.stderr or r.stdout).strip()[-300:]}
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {"ok": False, "snapshot": "", "files": [], "error": str(exc)}
|
||||
|
||||
latest = _latest()
|
||||
if not latest:
|
||||
return {"ok": False, "snapshot": "", "files": [], "error": "Kein Backup erzeugt"}
|
||||
return {
|
||||
"ok": True,
|
||||
"snapshot": _ts(latest),
|
||||
"files": _components(latest),
|
||||
"size_mb": round(latest.stat().st_size / 1_000_000, 2),
|
||||
}
|
||||
|
||||
|
||||
def list_backups() -> list[dict]:
|
||||
if not BACKUP_DIR.exists():
|
||||
return []
|
||||
out = []
|
||||
for p in sorted(BACKUP_DIR.glob("mc2-state-*.tar.gz"), reverse=True):
|
||||
out.append({
|
||||
"snapshot": _ts(p),
|
||||
"file": p.name,
|
||||
"size_mb": round(p.stat().st_size / 1_000_000, 2),
|
||||
})
|
||||
return out
|
||||
@@ -0,0 +1,143 @@
|
||||
"""
|
||||
Speicher-Budget & SETUP-BEWUSSTE ctx-Vergabe — EINE Quelle der Wahrheit.
|
||||
|
||||
Modelliert die auf der Box VERIFIZIERTE Residenz-Realität (llama-swap, Ein-Gruppen-
|
||||
Residenz, GTT ~124 GB):
|
||||
• Das Agent-Hirn (Rolle `hermes`) ist IMMER resident.
|
||||
• Weitere persist-Mitglieder (fast/vision) dürfen verdrängt werden, wenn ein großes
|
||||
on-demand-Modell lädt.
|
||||
Daraus folgt, wie viel Speicher NEBEN einem Zielmodell reserviert bleiben muss —
|
||||
und damit der größte Kontext, der wirklich passt (nicht nur für das Modell allein).
|
||||
|
||||
Vorher rechnete nur der Hirn-Wechsel (agent.py) setup-bewusst; die allgemeine
|
||||
ctx-Vergabe nahm den Gesamt-RAM in Isolation. Dieses Modul vereint beides.
|
||||
"""
|
||||
|
||||
import re
|
||||
|
||||
import psutil
|
||||
|
||||
from services.fit import (
|
||||
QUANT_BYTES_PER_PARAM,
|
||||
estimate_memory_gb,
|
||||
extract_params_b,
|
||||
max_ctx_in_budget,
|
||||
)
|
||||
|
||||
HEADROOM_GB = 4.0 # OS/Treiber/Fragmentierung
|
||||
|
||||
|
||||
def gtt_budget_gb() -> float:
|
||||
"""GPU-adressierbarer Speicher (GTT) in GB — die harte Obergrenze. Liest
|
||||
amdgpu.gttsize aus /proc/cmdline, sonst RAM minus OS-Reserve."""
|
||||
try:
|
||||
with open("/proc/cmdline") as f:
|
||||
m = re.search(r"amdgpu\.gttsize=(\d+)", f.read())
|
||||
if m:
|
||||
return round(int(m.group(1)) / 1024.0, 1)
|
||||
except Exception:
|
||||
pass
|
||||
return round(psutil.virtual_memory().total / (1024 ** 3) - 6.0, 1)
|
||||
|
||||
|
||||
def params_of_model(model: dict) -> float:
|
||||
"""Robuste Params (Mrd.) eines INSTALLIERTEN Modells: MAXIMUM aus Caps-Schätzung und
|
||||
Dateigröße. Deckt 'Coder-Next' ohne Größe im Namen (→ aus Datei) und Split-GGUFs
|
||||
(size_bytes = nur erster Teil → ignoriert) ab."""
|
||||
caps = model.get("capabilities") or {}
|
||||
quant = model.get("quant") or "Q4_K_M"
|
||||
bpp = QUANT_BYTES_PER_PARAM.get(quant.upper(), 0.55)
|
||||
size_gb = (model.get("size_bytes") or 0) / (1024 ** 3)
|
||||
pb_size = (size_gb / bpp) if size_gb > 1.0 else 0.0
|
||||
return max(float(caps.get("params_b") or 0), pb_size, 7.0)
|
||||
|
||||
|
||||
def footprint_gb(model: dict) -> float:
|
||||
"""Loaded-Footprint eines Modells = Gewichte + kalibrierter KV-Anteil (bei seinem
|
||||
aktuellen ctx)."""
|
||||
quant = model.get("quant") or "Q4_K_M"
|
||||
ctx = int(model.get("ctx") or 32768)
|
||||
bpp = QUANT_BYTES_PER_PARAM.get(quant.upper(), 0.55)
|
||||
size_gb = (model.get("size_bytes") or 0) / (1024 ** 3)
|
||||
pb = params_of_model(model)
|
||||
weights = max(pb * bpp, size_gb)
|
||||
kv = estimate_memory_gb(pb, quant, ctx) - pb * bpp
|
||||
return weights + max(kv, 0.0)
|
||||
|
||||
|
||||
def params_b_for(name: str) -> float:
|
||||
"""Parameter (Mrd.) für einen Modell-/Repo-Namen: KATALOG (echte Metadaten) zuerst,
|
||||
sonst Namens-Schätzung. Gemeinsam für Fit-Vorschau und ctx-Vergabe."""
|
||||
from services import catalog
|
||||
meta = catalog.meta_for_name(name) if name else None
|
||||
if meta and meta.get("total_params_b"):
|
||||
return float(meta["total_params_b"])
|
||||
return extract_params_b(name)
|
||||
|
||||
|
||||
def _coresident_members(groups: dict) -> set:
|
||||
"""Modelle, die GLEICHZEITIG warm sind: Mitglieder aller `swap:false`-Gruppen
|
||||
(Ko-Residenz, z.B. brains = Hirn+embed+vision). Ein Modell AUSSERHALB dieser Gruppen
|
||||
ist on-demand und verdrängt beim Laden die GANZE Gruppe — llama-swap swappt Gruppen
|
||||
(live verifiziert: heavy laden → brains-Gruppe komplett raus, heavy läuft allein)."""
|
||||
out: set = set()
|
||||
for g in (groups or {}).values():
|
||||
if isinstance(g, dict) and g.get("swap") is False:
|
||||
out.update(g.get("members") or [])
|
||||
return out
|
||||
|
||||
|
||||
def reserved_gb(role: str | None) -> dict:
|
||||
"""Speicher, der NEBEN einem Zielmodell der gegebenen Rolle resident bleibt — gemäß der
|
||||
VERIFIZIERTEN llama-swap-Gruppen-Swap-Semantik (NICHT der früheren Annahme „Hirn bleibt
|
||||
immer"). Nur die ko-residente `swap:false`-Gruppe läuft gemeinsam; ein on-demand-Modell
|
||||
verdrängt die Gruppe und läuft ALLEIN mit dem vollen GTT.
|
||||
|
||||
- Modell IN der Ko-Residenz-Gruppe (Hirn/embed/vision): koexistiert mit den ÜBRIGEN
|
||||
Gruppen-Mitgliedern → reserviert deren Summe.
|
||||
- Modell AUSSERHALB (heavy/coder/coder-lite/fast/scout): läuft allein (Gruppe wird beim
|
||||
Laden rausgeswappt) → reserviert NICHTS, darf den vollen GTT für Kontext nutzen.
|
||||
"""
|
||||
from services import llamaswap
|
||||
models = llamaswap.list_models()
|
||||
groups = llamaswap.list_groups()
|
||||
cores = _coresident_members(groups)
|
||||
brain = next((m for m in models if (m.get("role") == "hermes")), None)
|
||||
brain_gb = footprint_gb(brain) if brain else 0.0
|
||||
role = (role or "").strip().lower()
|
||||
|
||||
holder = next((m for m in models if (m.get("role") == role)), None) if role else None
|
||||
holder_name = holder["name"] if holder else None
|
||||
# Hirn (hermes) ist per Definition Teil der Ko-Residenz-Gruppe; sonst Gruppen-Mitgliedschaft prüfen.
|
||||
in_group = role == "hermes" or bool(holder_name and holder_name in cores)
|
||||
|
||||
if in_group:
|
||||
others = sum(footprint_gb(m) for m in models
|
||||
if m["name"] in cores and m["name"] != holder_name)
|
||||
return {"reserved_gb": others, "mode": "co-resident", "brain_gb": brain_gb}
|
||||
# on-demand: verdrängt die Ko-Residenz-Gruppe → läuft allein, voller GTT für Kontext.
|
||||
return {"reserved_gb": 0.0, "mode": "ondemand-alone", "brain_gb": brain_gb}
|
||||
|
||||
|
||||
def setup_aware_ctx(params_b: float, quant: str, role: str | None = None) -> dict:
|
||||
"""Größter Kontext, der für ein Modell (params_b/quant) der gegebenen Rolle NEBEN dem
|
||||
bestehenden Setup passt. Gibt ctx + die Budget-Herleitung zurück (für UI/Transparenz)."""
|
||||
gtt = gtt_budget_gb()
|
||||
r = reserved_gb(role)
|
||||
budget = max(gtt - r["reserved_gb"] - HEADROOM_GB, 0.0)
|
||||
ctx = max_ctx_in_budget(params_b, quant, budget)
|
||||
return {
|
||||
"ctx": ctx,
|
||||
"gtt_gb": gtt,
|
||||
"reserved_gb": round(r["reserved_gb"], 1),
|
||||
"budget_gb": round(budget, 1),
|
||||
"mode": r["mode"],
|
||||
}
|
||||
|
||||
|
||||
def setup_aware_ctx_for_model(model: dict) -> dict:
|
||||
"""Setup-bewusster Optimal-ctx für ein INSTALLIERTES Modell (aus seiner Rolle,
|
||||
Params & Quant). Für den 'Auto'-Button an der Modellkarte."""
|
||||
return setup_aware_ctx(params_of_model(model),
|
||||
model.get("quant") or "Q4_K_M",
|
||||
role=model.get("role"))
|
||||
@@ -0,0 +1,135 @@
|
||||
"""
|
||||
Modell-Capabilities — EINE Quelle der Wahrheit für Modell-Eigenschaften
|
||||
(MoE / Tools / Vision / Coder / Reasoning / Embedding / Kontext).
|
||||
Portiert aus Mission Control v1 (model_caps.py).
|
||||
|
||||
Quellen, geschichtet: GGUF-Header (offline, authoritativ) → cmd-Flags
|
||||
(--jinja/--mmproj) → HF-Block (tags + chat_template) → Familien-Fallback.
|
||||
Tool-Fähigkeit dreistufig: yes (bestätigt) | likely (Familie) | no.
|
||||
"""
|
||||
|
||||
import re
|
||||
import struct
|
||||
|
||||
from services.fit import extract_active_params_b, extract_params_b
|
||||
|
||||
_GGUF_FIXED = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
|
||||
|
||||
|
||||
def _read_gguf_meta(path: str) -> dict:
|
||||
"""Liest nur den GGUF-Metadaten-Header (architecture/context_length/expert_count/
|
||||
parameter_count). Bricht vor dem Tokenizer-Array ab → schnell, lädt NICHT das Modell."""
|
||||
out: dict = {}
|
||||
try:
|
||||
with open(path, "rb") as f:
|
||||
if f.read(4) != b"GGUF":
|
||||
return {}
|
||||
struct.unpack("<I", f.read(4))[0]
|
||||
f.read(8)
|
||||
kv = struct.unpack("<Q", f.read(8))[0]
|
||||
|
||||
def ru32() -> int: return struct.unpack("<I", f.read(4))[0]
|
||||
def ru64() -> int: return struct.unpack("<Q", f.read(8))[0]
|
||||
def rstr() -> str: return f.read(ru64()).decode("utf-8", "replace")
|
||||
|
||||
def rval(t: int):
|
||||
if t == 8: return rstr()
|
||||
if t == 0: return struct.unpack("<B", f.read(1))[0]
|
||||
if t == 1: return struct.unpack("<b", f.read(1))[0]
|
||||
if t == 2: return struct.unpack("<H", f.read(2))[0]
|
||||
if t == 3: return struct.unpack("<h", f.read(2))[0]
|
||||
if t == 4: return struct.unpack("<I", f.read(4))[0]
|
||||
if t == 5: return struct.unpack("<i", f.read(4))[0]
|
||||
if t == 6: return struct.unpack("<f", f.read(4))[0]
|
||||
if t == 7: return f.read(1) != b"\x00"
|
||||
if t == 10: return struct.unpack("<Q", f.read(8))[0]
|
||||
if t == 11: return struct.unpack("<q", f.read(8))[0]
|
||||
if t == 12: return struct.unpack("<d", f.read(8))[0]
|
||||
if t == 9:
|
||||
et = ru32(); cnt = ru64()
|
||||
if et == 8:
|
||||
for _ in range(cnt):
|
||||
f.seek(ru64(), 1)
|
||||
elif et == 9:
|
||||
for _ in range(cnt):
|
||||
rval(9)
|
||||
else:
|
||||
f.seek(cnt * _GGUF_FIXED.get(et, 0), 1)
|
||||
return None
|
||||
raise ValueError(f"unbekannter GGUF-Typ {t}")
|
||||
|
||||
want = {"architecture", "context_length", "expert_count", "parameter_count"}
|
||||
for _ in range(kv):
|
||||
key = rstr()
|
||||
t = ru32()
|
||||
if key == "tokenizer.ggml.tokens":
|
||||
break
|
||||
v = rval(t)
|
||||
short = key.split(".")[-1]
|
||||
if short in want and short not in out:
|
||||
out[short] = v
|
||||
except Exception:
|
||||
return out
|
||||
return out
|
||||
|
||||
|
||||
_TOOL_FAMILIES = (
|
||||
"qwen2.5", "qwen3", "qwen2", "hermes", "mistral", "mixtral", "devstral",
|
||||
"command-r", "command_r", "llama-3.1", "llama3.1", "llama-3.3", "llama-4", "llama4",
|
||||
"functionary", "watt", "firefunction", "granite", "glm-4", "glm-5", "ministral",
|
||||
)
|
||||
_REASON_KW = (
|
||||
"-r1", "deepseek-r1", "qwq", "magistral", "-think", "thinking", "-o1",
|
||||
"gpt-oss", "reasoning", "exaone-deep", "phi-4-reasoning", "phi-4-mini-reasoning",
|
||||
)
|
||||
_CODE_KW = ("coder", "-code", "code-", "codestral", "starcoder", "deepseek-coder")
|
||||
_VISION_KW = ("-vl", "vision", "llava", "pixtral", "multimodal", "-mm-", "qwen3vl", "qwen2-vl")
|
||||
_EMBED_KW = ("bge", "e5-", "gte-", "nomic-embed", "embed")
|
||||
_MOE_ARCH = ("moe", "mixtral", "deepseek2", "deepseek3", "llama4", "qwen3moe", "grok")
|
||||
|
||||
|
||||
def capabilities(name: str = "", cmd: str = "", gguf_path: str = "", hf: dict | None = None) -> dict:
|
||||
"""Capability-Tag-Set für ein Modell. Alle Quellen optional — nutzt, was da ist."""
|
||||
low = (name or "").lower()
|
||||
cmdl = (cmd or "").lower()
|
||||
hf = hf or {}
|
||||
|
||||
meta = _read_gguf_meta(gguf_path) if gguf_path else {}
|
||||
arch = str(meta.get("architecture") or hf.get("architecture") or "").lower()
|
||||
tags = [str(t).lower() for t in (hf.get("tags") or [])]
|
||||
chat_tpl = str(hf.get("chat_template") or "")
|
||||
|
||||
expert_count = int(meta.get("expert_count") or 0)
|
||||
moe = (
|
||||
expert_count > 1
|
||||
or any(a in arch for a in _MOE_ARCH)
|
||||
or bool(re.search(r"\d+x\d+\.?\d*b", low))
|
||||
or bool(re.search(r"a\d+\.?\d*b", low))
|
||||
)
|
||||
active_b = extract_active_params_b(name)
|
||||
|
||||
pcount = int(meta.get("parameter_count") or 0)
|
||||
params_b = round(pcount / 1e9, 1) if pcount else extract_params_b(name)
|
||||
ctx = meta.get("context_length")
|
||||
if not ctx:
|
||||
m = re.search(r"-(?:c|-ctx-size)\s+(\d+)", cmdl)
|
||||
ctx = int(m.group(1)) if m else None
|
||||
|
||||
# Cap context length at 131072 for Qwen / Hermes models to prevent reporting scaled RoPE context of 256k+ which might OOM or be unstable.
|
||||
if ctx and ctx > 131072 and ("qwen" in low or "hermes" in low):
|
||||
ctx = 131072
|
||||
|
||||
tool_confirmed = "--jinja" in cmdl or "tool_call" in chat_tpl or "<tools>" in chat_tpl
|
||||
tool_family = any(fam in low for fam in _TOOL_FAMILIES) or "function-calling" in tags
|
||||
tools = "yes" if tool_confirmed else ("likely" if tool_family else "no")
|
||||
|
||||
vision = "--mmproj" in cmdl or "vl" in arch or "clip" in arch or any(k in low for k in _VISION_KW)
|
||||
coder = any(k in low for k in _CODE_KW)
|
||||
reasoning = any(k in low for k in _REASON_KW) or "reasoning" in tags
|
||||
embedding = "bert" in arch or any(k in low for k in _EMBED_KW)
|
||||
|
||||
return {
|
||||
"moe": moe, "active_b": active_b, "tools": tools, "vision": vision,
|
||||
"coder": coder, "reasoning": reasoning, "embedding": embedding,
|
||||
"ctx": ctx, "params_b": params_b or None, "arch": arch or None,
|
||||
}
|
||||
@@ -0,0 +1,115 @@
|
||||
"""
|
||||
Kuratierter Modell-Katalog ("Cookbook", inspiriert von Odysseus): EINE Quelle der
|
||||
Wahrheit für KORREKTE Metadaten (total/active params, moe, generation) statt
|
||||
Namens-Raterei. Macht Empfehlung + Upgrade-Erkennung präzise und MoE-bewusst
|
||||
für die bandbreiten-limitierte Strix-Halo-Box.
|
||||
|
||||
Daten: backend/models_catalog.json. Fällt sanft aus (leerer Katalog), wenn die
|
||||
Datei fehlt → discover nutzt dann nur die HF-Dynamik.
|
||||
"""
|
||||
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import re
|
||||
|
||||
from services.fit import estimate_memory_gb, estimate_speed
|
||||
|
||||
_CATALOG_PATH = os.path.join(os.path.dirname(__file__), "..", "models_catalog.json")
|
||||
_cache: dict = {"mtime": 0.0, "models": []}
|
||||
|
||||
|
||||
def _load() -> list[dict]:
|
||||
try:
|
||||
mt = os.path.getmtime(_CATALOG_PATH)
|
||||
if mt != _cache["mtime"]:
|
||||
with open(_CATALOG_PATH, encoding="utf-8") as f:
|
||||
data = json.load(f) or {}
|
||||
_cache.update(mtime=mt, models=[m for m in data.get("models", []) if m.get("name")])
|
||||
except (OSError, ValueError):
|
||||
_cache.update(mtime=0.0, models=[])
|
||||
return _cache["models"]
|
||||
|
||||
|
||||
def _norm(name: str) -> str:
|
||||
"""Vergleichs-Stamm: kleingeschrieben, Org-Prefix/Quant/GGUF/Split entfernt."""
|
||||
s = (name or "").lower().split("/")[-1]
|
||||
s = re.sub(r"\.gguf$", "", s)
|
||||
s = re.sub(r"-\d+-of-\d+$", "", s)
|
||||
s = re.sub(r"[-_](ud-)?(i?q\d[\w]*|f16|bf16|fp16|f32|mxfp4)$", "", s)
|
||||
return s.strip("-_ ")
|
||||
|
||||
|
||||
def entries() -> list[dict]:
|
||||
return list(_load())
|
||||
|
||||
|
||||
def entries_for_role(role: str) -> list[dict]:
|
||||
return [e for e in _load() if e.get("role") == role]
|
||||
|
||||
|
||||
def meta_for_name(name: str) -> dict | None:
|
||||
"""Katalog-Metadaten zu einem Modell(namen) — matcht lokalen Namen ODER HF-Repo."""
|
||||
n = _norm(name)
|
||||
if not n:
|
||||
return None
|
||||
for e in _load():
|
||||
cand = {_norm(e.get("name", "")), _norm(e.get("repo", ""))}
|
||||
if n in cand or any(c and (c in n or n in c) for c in cand):
|
||||
return e
|
||||
return None
|
||||
|
||||
|
||||
def fit_of(e: dict, ram_gb: float) -> dict:
|
||||
"""Hardware-Fit eines Katalog-Eintrags (MoE-bewusst über active_params_b)."""
|
||||
total = float(e.get("total_params_b") or 7)
|
||||
active = float(e.get("active_params_b") or total)
|
||||
quant = e.get("quant") or "Q4_K_M"
|
||||
ctx = int(e.get("ctx") or 32768)
|
||||
req_gb = estimate_memory_gb(total, quant, ctx)
|
||||
tps = estimate_speed(req_gb, ram_gb, (active / total) if total else 1.0)
|
||||
usable = max(ram_gb - 4.0, 0)
|
||||
if req_gb > usable:
|
||||
level, text = "too_tight", "Zu groß (OOM)"
|
||||
elif req_gb > usable * 0.8:
|
||||
level, text = "marginal", "Könnte knapp werden"
|
||||
else:
|
||||
level, text = "perfect", "Passt perfekt"
|
||||
return {"level": level, "text": text, "req_gb": round(req_gb, 1), "tps": round(tps, 0)}
|
||||
|
||||
|
||||
def stack_score(e: dict, ram_gb: float) -> float:
|
||||
"""Score für DIESE Hardware: muss passen, dann Wissen (total params) + Tempo
|
||||
(tps — belohnt MoE durch niedrige aktive Params automatisch). Bandbreiten-Box
|
||||
→ MoE gewinnt bei vergleichbarem Wissen gegen dense."""
|
||||
fit = fit_of(e, ram_gb)
|
||||
if fit["level"] == "too_tight":
|
||||
return -100.0
|
||||
total = float(e.get("total_params_b") or 7)
|
||||
fit_bonus = 3.0 if fit["level"] == "perfect" else 1.0
|
||||
knowledge = math.log2(total + 1) / 8.0 # ~0..1 (bis ~256B)
|
||||
speed = min((fit["tps"] or 0) / 80.0, 1.0) # normalisiert; MoE = hohe tps
|
||||
return fit_bonus + 1.2 * knowledge + 1.0 * speed
|
||||
|
||||
|
||||
def to_model_dict(e: dict, ram_gb: float) -> dict:
|
||||
"""Katalog-Eintrag → discover-kompatibles Modell-Dict (echte Metadaten)."""
|
||||
total = float(e.get("total_params_b") or 7)
|
||||
active = e.get("active_params_b")
|
||||
repo = e.get("repo") or e.get("name")
|
||||
role = e.get("role")
|
||||
caps = {
|
||||
"moe": bool(e.get("moe")), "active_b": active,
|
||||
"tools": "yes" if e.get("tools") else "no",
|
||||
"vision": bool(e.get("vision")), "coder": role == "coder",
|
||||
"reasoning": role == "heavy", "embedding": False,
|
||||
"ctx": e.get("ctx"), "params_b": total, "arch": e.get("family"),
|
||||
}
|
||||
return {
|
||||
"name": e.get("name"), "author": repo.split("/")[0] if "/" in repo else "catalog",
|
||||
"repo": repo, "role": role, "params_b": total, "active_b": active,
|
||||
"moe": bool(e.get("moe")), "generation": e.get("generation"),
|
||||
"family": e.get("family"), "quant": e.get("quant") or "Q4_K_M",
|
||||
"tags": ["catalog"], "downloads": 0, "fit": fit_of(e, ram_gb),
|
||||
"optimal_ctx": int(e.get("ctx") or 32768), "caps": caps, "curated": True,
|
||||
}
|
||||
@@ -0,0 +1,160 @@
|
||||
"""
|
||||
Connect: erzeugt saubere, getestete Konfig-Snippets für IDEs/Agenten auf dem
|
||||
LOKALEN PC (separate Maschine im LAN). Alle zeigen auf den **Gateway** der Box
|
||||
(Lanes `coding`/`chat`, Cockpit-Port :9001/v1) + den **Shared-Memory-MCP** (MC :9001).
|
||||
|
||||
Wichtig: Host ist die LAN-IP der Box (NICHT eine Proxy-Domain) — das war in v1
|
||||
die häufigste Fehlerquelle. Der Aufrufer übergibt den Host explizit.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import httpx
|
||||
|
||||
from config import LLAMA_SWAP_URL, MEM0_SERVICE_URL, PORT
|
||||
|
||||
DEFAULT_HOST = "192.168.178.151"
|
||||
|
||||
# IDEs bekommen NUR die 'coding'-Lane zu sehen — die Lane routet intern selbst auf das
|
||||
# passende Modell (Coder/Heavy/…). Ein einziger Eintrag, kein manuelles Modell-Wählen mehr.
|
||||
IDE_MODEL = "coding"
|
||||
|
||||
|
||||
def _gw(host: str) -> str:
|
||||
# Eingebauter Gateway: MC2 serviert /v1 selbst (gleicher Port wie das Cockpit).
|
||||
return f"http://{host}:{PORT}/v1"
|
||||
|
||||
|
||||
def build_snippets(host: str = DEFAULT_HOST,
|
||||
mcp_script_path: str = r"F:\\Coding Stuff\\mission-control-2\\mcp\\mcp_memory.py",
|
||||
mcp_python: str = "python") -> dict:
|
||||
gw = _gw(host)
|
||||
mc_url = f"http://{host}:{PORT}"
|
||||
|
||||
cline = json.dumps({
|
||||
"apiProvider": "openai",
|
||||
"openAiBaseUrl": gw,
|
||||
"openAiApiKey": "local",
|
||||
"openAiModelId": "coding",
|
||||
}, indent=2)
|
||||
|
||||
opencode = json.dumps({
|
||||
"provider": {
|
||||
"bosgame": {
|
||||
"npm": "@ai-sdk/openai-compatible",
|
||||
"name": "Bosgame Gateway",
|
||||
"options": {"baseURL": gw, "apiKey": "local"},
|
||||
"models": {IDE_MODEL: {"name": IDE_MODEL}},
|
||||
}
|
||||
}
|
||||
}, indent=2)
|
||||
|
||||
cursor = json.dumps({
|
||||
"Base URL": gw,
|
||||
"API Key": "local",
|
||||
"Active Model": "coding"
|
||||
}, indent=2)
|
||||
|
||||
zed = json.dumps({
|
||||
"language_models": {
|
||||
"openai_compatible": {
|
||||
"bosgame": {
|
||||
"api_url": gw,
|
||||
"available_models": [
|
||||
{"name": IDE_MODEL, "display_name": IDE_MODEL, "max_tokens": 131072,
|
||||
"capabilities": {"tools": True}}
|
||||
],
|
||||
}
|
||||
}
|
||||
},
|
||||
"assistant": {
|
||||
"default_model": {
|
||||
"provider": "openai_compatible",
|
||||
"model": IDE_MODEL
|
||||
}
|
||||
}
|
||||
}, indent=2)
|
||||
|
||||
cont = json.dumps({
|
||||
"models": [
|
||||
{"title": f"Bosgame / {IDE_MODEL}", "provider": "openai", "model": IDE_MODEL,
|
||||
"apiBase": gw, "apiKey": "local"}
|
||||
]
|
||||
}, indent=2)
|
||||
|
||||
# Claude Code spricht das Anthropic-Format; der Gateway ist OpenAI-kompatibel und
|
||||
# bietet KEIN /v1/messages (verifiziert). Daher braucht es einen kleinen Übersetzer
|
||||
# (Anthropic ⇄ OpenAI) als Aufsatz. Die env-Vars sind Claude Codes echte Schnittstelle.
|
||||
claude_code = (
|
||||
f"# Claude Code spricht das Anthropic-Format — der Gateway ist OpenAI-kompatibel ({gw})\n"
|
||||
f"# und hat kein /v1/messages. Dazwischen muss ein Übersetzer (Anthropic ⇄ OpenAI) laufen:\n"
|
||||
f"# • claude-code-router (leichtgewichtig, npm)\n"
|
||||
f"# • oder LiteLLM mit /v1/messages-Bridge\n"
|
||||
f"# Den Übersetzer auf den Gateway zeigen lassen: baseURL={gw}, model=coding, apiKey=local.\n"
|
||||
f"# Dann Claude Code auf den lokalen Übersetzer richten (Beispiel-Port 3456):\n"
|
||||
f"\n"
|
||||
f'export ANTHROPIC_BASE_URL="http://localhost:3456"\n'
|
||||
f'export ANTHROPIC_AUTH_TOKEN="local"\n'
|
||||
f'export ANTHROPIC_MODEL="coding"'
|
||||
)
|
||||
|
||||
memory_mcp = json.dumps({
|
||||
"mcpServers": {
|
||||
"mission-control-memory": {
|
||||
"command": mcp_python,
|
||||
"args": [mcp_script_path],
|
||||
"env": {"MC_URL": mc_url},
|
||||
}
|
||||
}
|
||||
}, indent=2)
|
||||
|
||||
return {
|
||||
"host": host,
|
||||
"gateway_url": gw,
|
||||
"mc_url": mc_url,
|
||||
# Leitung 1 — das MODELL. Alle Snippets zeigen auf den OpenAI-kompatiblen Gateway.
|
||||
"tools": {
|
||||
"cline": {"label": "Roo Code / Cline", "lang": "json", "snippet": cline,
|
||||
"note": "OpenAI-Provider → Gateway. Nur Lane 'coding' — die Box routet intern selbst."},
|
||||
"cursor": {"label": "Cursor", "lang": "json", "snippet": cursor,
|
||||
"note": "Einstellungen ➔ Models ➔ OpenAI API key + Base URL."},
|
||||
"opencode": {"label": "OpenCode", "lang": "jsonc", "snippet": opencode,
|
||||
"note": "Datei opencode.jsonc, Key 'provider'."},
|
||||
"zed": {"label": "Zed", "lang": "json", "snippet": zed,
|
||||
"note": "settings.json → language_models.openai_compatible."},
|
||||
"continue": {"label": "Continue", "lang": "json", "snippet": cont,
|
||||
"note": "~/.continue/config.json (oder config.yaml mit identischen Keys)."},
|
||||
"claude_code": {"label": "Claude Code", "lang": "bash", "snippet": claude_code,
|
||||
"note": "Braucht einen Anthropic⇄OpenAI-Übersetzer vor dem Gateway."},
|
||||
},
|
||||
# Leitung 2 — das GEDÄCHTNIS. Separater MCP-Server, gilt zusätzlich zu jedem Tool oben.
|
||||
"memory": {"label": "Shared Memory (MCP)", "lang": "json", "snippet": memory_mcp,
|
||||
"note": "Eigene Leitung: MCP-Block für jedes MCP-fähige Tool. mcp_memory.py muss lokal liegen."},
|
||||
}
|
||||
|
||||
|
||||
def check_health() -> dict:
|
||||
"""Live-Erreichbarkeit der beiden Leitungen, aus Sicht der Box:
|
||||
Leitung 1 = Gateway/Engine (llama-swap), Leitung 2 = Gedächtnis-Sidecar (Mem0)."""
|
||||
gateway = {"ok": False, "detail": "nicht erreichbar"}
|
||||
try:
|
||||
with httpx.Client(timeout=3.0) as c:
|
||||
r = c.get(f"{LLAMA_SWAP_URL}/v1/models")
|
||||
if r.status_code == 200:
|
||||
n = len(r.json().get("data", []))
|
||||
gateway = {"ok": True, "detail": f"{n} Modelle verfügbar" if n else "bereit"}
|
||||
else:
|
||||
gateway = {"ok": False, "detail": f"HTTP {r.status_code}"}
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
memory = {"ok": False, "detail": "nicht erreichbar"}
|
||||
try:
|
||||
with httpx.Client(timeout=3.0) as c:
|
||||
r = c.get(f"{MEM0_SERVICE_URL}/health")
|
||||
memory = ({"ok": True, "detail": "bereit"} if r.status_code == 200
|
||||
else {"ok": False, "detail": f"HTTP {r.status_code}"})
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
return {"gateway": gateway, "memory": memory}
|
||||
@@ -0,0 +1,179 @@
|
||||
"""
|
||||
Automatische Modell-Entdeckung ("aktuell beste Modelle"): fragt vertrauenswürdige
|
||||
HF-Orgs live ab, kategorisiert per Stichwort, rankt nach Hardware-Fit + Beliebtheit
|
||||
und cached. Portiert aus Mission Control v1 (cookbook.py-Discover).
|
||||
|
||||
Wichtig (Greenfield-Fix gegen v1): EIN gemeinsamer Ranking-Helfer `rank_runnable`
|
||||
ist die Quelle der Wahrheit — sowohl die „beste Empfehlung" je Kategorie als auch
|
||||
spätere Auto-Setups nutzen ihn, damit sie nie auseinanderlaufen.
|
||||
"""
|
||||
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import time
|
||||
from datetime import datetime
|
||||
|
||||
import httpx
|
||||
|
||||
import logging
|
||||
|
||||
from config import DISCOVER_CACHE_PATH, DISCOVER_TTL
|
||||
from services import catalog
|
||||
from services.caps import capabilities
|
||||
from services.fit import evaluate_fit, extract_params_b, max_ctx_for
|
||||
from services.sources import CATEGORIES, SKIP_TOKENS, TRUSTED_AUTHORS
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_FIT_ORDER = {"perfect": 0, "marginal": 1, "too_tight": 2}
|
||||
|
||||
|
||||
def _categorize(repo_id: str) -> str:
|
||||
low = repo_id.lower()
|
||||
for cat in CATEGORIES:
|
||||
if any(k in low for k in cat["kw"]):
|
||||
return cat["role"]
|
||||
return "scout"
|
||||
|
||||
|
||||
def _fetch_author_models(author: str) -> list:
|
||||
url = (f"https://huggingface.co/api/models?author={author}"
|
||||
f"&filter=gguf&sort=downloads&direction=-1&limit=40")
|
||||
try:
|
||||
with httpx.Client(timeout=12.0) as c:
|
||||
data = c.get(url).json()
|
||||
return data if isinstance(data, list) else []
|
||||
except Exception:
|
||||
log.debug("discover: Abfrage für Autor %s fehlgeschlagen", author, exc_info=True)
|
||||
return []
|
||||
|
||||
|
||||
def _age_days(last_modified, now_ts: float) -> float:
|
||||
"""Alter eines HF-Modells in Tagen (lastModified ISO). Unbekannt → ~1.5 Jahre."""
|
||||
if not last_modified:
|
||||
return 540.0
|
||||
try:
|
||||
dt = datetime.fromisoformat(str(last_modified).replace("Z", "+00:00"))
|
||||
return max((now_ts - dt.timestamp()) / 86400.0, 0.0)
|
||||
except Exception:
|
||||
return 540.0
|
||||
|
||||
|
||||
def _score(m: dict, now_ts: float) -> float:
|
||||
"""Zukunftssicherer Rang-Score für DIESE Hardware. Kombiniert:
|
||||
- Fit: perfect dominiert (Bonus 3.0 > Summe der übrigen Terme → passt-komfortabel zuerst),
|
||||
- Recency: neuere Generationen bevorzugt (Halbwertszeit ~9 Monate über lastModified),
|
||||
- Capability: mehr Parameter (log-skaliert),
|
||||
- Popularity: Downloads (log-skaliert).
|
||||
So gewinnt bei vergleichbarer Größe die NEUERE Generation (z.B. Qwen3-Coder vor
|
||||
Qwen2.5-Coder), ohne dass kleine Populär-Modelle große verdrängen."""
|
||||
fit_bonus = 3.0 if m["fit"]["level"] == "perfect" else 0.0
|
||||
recency = 0.5 ** (_age_days(m.get("lastModified"), now_ts) / 270.0)
|
||||
cap = math.log2(max(float(m.get("params_b") or 1.0), 1.0) + 1.0) / 8.0
|
||||
pop = math.log10(float(m.get("downloads") or 0) + 1.0) / 7.0
|
||||
return fit_bonus + 1.2 * recency + 1.2 * cap + 0.5 * pop
|
||||
|
||||
|
||||
def rank_runnable(models: list[dict]) -> list[dict]:
|
||||
"""EINE Quelle der Wahrheit fürs Ranking lauffähiger Modelle für DIESE Hardware.
|
||||
Nur was passt (too_tight fliegt raus), dann nach `_score` (Fit + Recency + Capability
|
||||
+ Popularity). Bevorzugt neuere, fähige Modelle → zukunftssicher; „Modelle finden"
|
||||
schlägt nie ein Downgrade vor (Downgrade-Sperre zusätzlich in maintenance)."""
|
||||
now_ts = time.time()
|
||||
return sorted(
|
||||
[m for m in models if m["fit"]["level"] != "too_tight"],
|
||||
key=lambda m: -_score(m, now_ts),
|
||||
)
|
||||
|
||||
|
||||
def refresh_discover(ram_gb: float) -> dict:
|
||||
"""Quellen live abfragen, kategorisieren, ranken, cachen. Wirft nur, wenn KEINE
|
||||
Quelle erreichbar war."""
|
||||
raw, seen, ok = [], set(), 0
|
||||
for author in TRUSTED_AUTHORS:
|
||||
models = _fetch_author_models(author)
|
||||
if models:
|
||||
ok += 1
|
||||
for m in models:
|
||||
rid = m.get("id")
|
||||
if not rid or rid in seen:
|
||||
continue
|
||||
seen.add(rid)
|
||||
raw.append(m)
|
||||
if ok == 0 and not catalog.entries():
|
||||
raise RuntimeError("Keine Quelle erreichbar.")
|
||||
|
||||
by_cat: dict[str, list] = {c["role"]: [] for c in CATEGORIES}
|
||||
for m in raw:
|
||||
rid = m["id"]
|
||||
low = rid.lower()
|
||||
if any(tok in low for tok in SKIP_TOKENS):
|
||||
continue
|
||||
role = _categorize(rid)
|
||||
params_b = extract_params_b(rid)
|
||||
quant = "Q4_K_M" # Referenz-Quant für die Fit-Einschätzung
|
||||
fit = evaluate_fit(params_b, quant, 8192, ram_gb, name=rid)
|
||||
tags = [str(t) for t in (m.get("tags") or [])]
|
||||
by_cat[role].append({
|
||||
"name": rid.split("/")[-1], "author": rid.split("/")[0], "repo": rid,
|
||||
"role": role, "params_b": params_b, "quant": quant, "tags": tags,
|
||||
"downloads": int(m.get("downloads") or 0), "likes": int(m.get("likes") or 0),
|
||||
"lastModified": m.get("lastModified"),
|
||||
"fit": fit, "optimal_ctx": max_ctx_for(params_b, quant, ram_gb),
|
||||
"caps": capabilities(name=rid, hf={"tags": tags}),
|
||||
})
|
||||
|
||||
cats = []
|
||||
for c in CATEGORIES:
|
||||
role = c["role"]
|
||||
# 1) KATALOG zuerst (kuratierte, korrekte Metadaten, MoE-bewusst gerankt) —
|
||||
# macht die Empfehlung präzise statt Namens-Raterei.
|
||||
cat_entries = sorted(catalog.entries_for_role(role),
|
||||
key=lambda e: -catalog.stack_score(e, ram_gb))
|
||||
cat_models = [m for m in (catalog.to_model_dict(e, ram_gb) for e in cat_entries)
|
||||
if m["fit"]["level"] != "too_tight"]
|
||||
# 2) HF-Dynamik als Ergänzung (nicht-kuratierte Funde), dedupliziert.
|
||||
hf_ranked = rank_runnable(by_cat[role])
|
||||
seen = {catalog._norm(m["repo"]) for m in cat_models}
|
||||
extra = [h for h in hf_ranked if catalog._norm(h["repo"]) not in seen]
|
||||
combined = cat_models + extra
|
||||
if combined:
|
||||
cats.append({
|
||||
"role": role, "title": c["title"], "icon": c["icon"],
|
||||
"models": combined[:6],
|
||||
# Empfehlung = bester KURATIERTER Eintrag, sonst beste HF-Fundstelle.
|
||||
"recommended": (cat_models[0]["repo"] if cat_models
|
||||
else (hf_ranked[0]["repo"] if hf_ranked else None)),
|
||||
})
|
||||
|
||||
data = {"updated": time.time(), "categories": cats}
|
||||
try:
|
||||
DISCOVER_CACHE_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = DISCOVER_CACHE_PATH.with_name(DISCOVER_CACHE_PATH.name + ".tmp")
|
||||
tmp.write_text(json.dumps(data, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
os.replace(tmp, DISCOVER_CACHE_PATH)
|
||||
except Exception:
|
||||
log.debug("discover: Cache-Schreiben fehlgeschlagen (nur Beschleunigung)", exc_info=True)
|
||||
return data
|
||||
|
||||
|
||||
def load_discover() -> dict | None:
|
||||
try:
|
||||
if DISCOVER_CACHE_PATH.exists():
|
||||
return json.loads(DISCOVER_CACHE_PATH.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
log.debug("discover: Cache-Lesen fehlgeschlagen", exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def safe_discover(ram_gb: float) -> dict | None:
|
||||
"""Aus Cache (wenn frisch) oder live; wirft nie — None wenn nichts da."""
|
||||
cached = load_discover()
|
||||
if cached and (time.time() - cached.get("updated", 0) < DISCOVER_TTL):
|
||||
return cached
|
||||
try:
|
||||
return refresh_discover(ram_gb)
|
||||
except Exception:
|
||||
log.warning("discover: Live-Refresh fehlgeschlagen, nutze Cache", exc_info=True)
|
||||
return cached
|
||||
@@ -0,0 +1,106 @@
|
||||
"""
|
||||
Hardware-Fit-Mathe (VRAM/RAM, tps-Schätzung) für APUs mit Unified Memory
|
||||
(Bosgame M5 / Strix Halo). Portiert aus Mission Control v1 (hw_math.py).
|
||||
"""
|
||||
|
||||
import re
|
||||
|
||||
# Bytes pro Parameter je GGUF-Quant (Annahme).
|
||||
QUANT_BYTES_PER_PARAM = {
|
||||
"Q2_K": 0.35, "Q3_K_S": 0.38, "Q3_K_M": 0.42, "Q3_K_L": 0.45,
|
||||
"Q4_0": 0.50, "Q4_1": 0.55, "Q4_K_S": 0.50, "Q4_K_M": 0.55,
|
||||
"Q5_0": 0.62, "Q5_1": 0.68, "Q5_K_S": 0.62, "Q5_K_M": 0.65,
|
||||
"Q6_K": 0.75, "Q8_0": 1.00, "F16": 2.00, "BF16": 2.00,
|
||||
"MXFP4": 0.55, "FP8": 1.05, "AWQ": 0.55,
|
||||
}
|
||||
|
||||
|
||||
def estimate_memory_gb(params_b: float, quant: str, ctx: int) -> float:
|
||||
"""Geschätzter Speicherbedarf in GB (Gewichte + Kontext-KV).
|
||||
KV-Cache skaliert NICHT linear mit den Gesamt-Parametern (er hängt an
|
||||
Layern × KV-Heads, gedämpft durch GQA) → sqrt-Skalierung, kalibriert am
|
||||
gemessenen Punkt Hermes-4-14B @ 128K ≈ 19 GB KV."""
|
||||
bpp = QUANT_BYTES_PER_PARAM.get(quant.upper(), 0.65)
|
||||
weights = params_b * bpp
|
||||
context_vram = (ctx / 8192) * (max(params_b, 7) / 7) ** 0.5 * 0.84
|
||||
return weights + context_vram
|
||||
|
||||
|
||||
def extract_active_params_b(name: str) -> float | None:
|
||||
"""Aktive Parameter bei MoE ('30B-A3B' → 3.0). None bei Dense."""
|
||||
m = re.search(r"(?<![a-zA-Z])a(\d+(?:\.\d+)?)b\b", name.lower())
|
||||
return float(m.group(1)) if m else None
|
||||
|
||||
|
||||
def estimate_speed(req_gb: float, sys_ram_gb: float, moe_active_ratio: float = 1.0) -> float:
|
||||
"""Geschätzte t/s anhand der ~273 GB/s Bandbreite der APU.
|
||||
moe_active_ratio = aktive/gesamt Params; < 1 bei MoE."""
|
||||
bw = 273 if sys_ram_gb > 8 else 70
|
||||
if req_gb <= 0:
|
||||
return 0.0
|
||||
raw_tps = (bw / req_gb) * 0.55
|
||||
if moe_active_ratio < 0.8:
|
||||
raw_tps *= (1.0 / moe_active_ratio) ** 0.5
|
||||
return raw_tps
|
||||
|
||||
|
||||
def evaluate_fit(params_b: float, quant: str, ctx: int, sys_ram_gb: float, name: str = "") -> dict:
|
||||
"""Fit für ein Shared-Memory-System (APU). name → MoE-Erkennung (optional)."""
|
||||
req_gb = estimate_memory_gb(params_b, quant, ctx)
|
||||
active_b = extract_active_params_b(name) if name else None
|
||||
moe_ratio = (active_b / params_b) if (active_b and params_b > 0) else 1.0
|
||||
tps = estimate_speed(req_gb, sys_ram_gb, moe_ratio)
|
||||
usable_ram = max(sys_ram_gb - 4.0, 0)
|
||||
if req_gb > usable_ram:
|
||||
fit_level, text = "too_tight", "Zu groß (OOM)"
|
||||
elif req_gb > usable_ram * 0.8:
|
||||
fit_level, text = "marginal", "Könnte knapp werden"
|
||||
else:
|
||||
fit_level, text = "perfect", "Passt perfekt"
|
||||
return {"level": fit_level, "text": text, "req_gb": round(req_gb, 1), "tps": round(tps, 0)}
|
||||
|
||||
|
||||
def extract_params_b(name: str) -> float:
|
||||
"""Parametergröße (Mrd.) aus Repo-/Dateiname. 8x7B (MoE) → 56."""
|
||||
moe = re.search(r"(\d+)x(\d+(?:\.\d+)?)[bB]", name)
|
||||
if moe:
|
||||
return float(moe.group(1)) * float(moe.group(2))
|
||||
m = re.search(r"(\d+(?:\.\d+)?)[bB](?![a-zA-Z])", name)
|
||||
return float(m.group(1)) if m else 7.0
|
||||
|
||||
|
||||
_NICE_CTX = [2048, 4096, 8192, 16384, 32768, 49152, 65536, 98304, 131072]
|
||||
|
||||
|
||||
def max_ctx_in_budget(params_b: float, quant: str, budget_gb: float) -> int:
|
||||
"""Größter 'schöner' Kontext, dessen Gewichte + KV in budget_gb passen.
|
||||
Budget-basierter Kern → wird von der setup-bewussten ctx-Vergabe
|
||||
(services.budget) mit dem ECHTEN freien Budget gefüttert."""
|
||||
bpp = QUANT_BYTES_PER_PARAM.get(quant.upper(), 0.65)
|
||||
weights = params_b * bpp
|
||||
ctx_budget = budget_gb - weights
|
||||
if ctx_budget <= 0:
|
||||
return 2048
|
||||
# KV pro 8k — EXAKTE Inverse von estimate_memory_gb (sqrt, kalibriert an
|
||||
# Hermes-14B@128K≈19GB). Vorher linear → für große Modelle viel zu konservativ.
|
||||
per_8k = (max(params_b, 7) / 7) ** 0.5 * 0.84
|
||||
raw_ctx = (ctx_budget / per_8k) * 8192
|
||||
best = _NICE_CTX[0]
|
||||
for c in _NICE_CTX:
|
||||
if c <= raw_ctx:
|
||||
best = c
|
||||
return best
|
||||
|
||||
|
||||
def max_ctx_for(params_b: float, quant: str, sys_ram_gb: float) -> int:
|
||||
"""Roh-Obergrenze: größter Kontext für dieses Modell ALLEIN gegen den
|
||||
Gesamt-RAM (80 % nutzbar). Ignoriert bewusst das übrige Setup —
|
||||
setup-bewusst rechnet services.budget.setup_aware_ctx."""
|
||||
return max_ctx_in_budget(params_b, quant, max(sys_ram_gb - 4.0, 0) * 0.8)
|
||||
|
||||
|
||||
def recommend_ctx(params_b: float, quant: str, sys_ram_gb: float) -> dict:
|
||||
ctx = max_ctx_for(params_b, quant, sys_ram_gb)
|
||||
k = ctx // 1024
|
||||
return {"ctx": ctx, "k": k,
|
||||
"note": f"Bis ~{k}k Kontext passt komfortabel auf deine Hardware ({round(sys_ram_gb)} GB)."}
|
||||
@@ -0,0 +1,46 @@
|
||||
"""
|
||||
Routing-Gateway-Status (eingebauter Modus). MC2 IST der Gateway: serviert
|
||||
`/v1/*` mit `model: auto`-Komplexitäts-Routing vor llama-swap. Kein externer
|
||||
LiteLLM-Dienst nötig (baut auf Python 3.14 nicht); bleibt später austauschbar.
|
||||
"""
|
||||
|
||||
from config import PORT
|
||||
from services.llamaswap import engine_reachable
|
||||
from services.routing_policy import load_policy
|
||||
|
||||
|
||||
def routing_summary() -> dict:
|
||||
p = load_policy()
|
||||
coding_default = p["coder_lite"] or p["coder"]
|
||||
return {
|
||||
"mode": "builtin",
|
||||
"endpoint": f":{PORT}/v1 (OpenAI-kompatibel)",
|
||||
# Virtuelle Lanes, die Clients/IDEs als „Modell" wählen (Router pickt das echte Alias).
|
||||
"lanes": [
|
||||
{
|
||||
"name": "chat",
|
||||
"aka": "auto",
|
||||
"target": f"{p['fast']} ↔ {p['heavy']} (nach Komplexität)",
|
||||
"threshold_chars": p["heavy_chars"],
|
||||
},
|
||||
{
|
||||
"name": "coding",
|
||||
"target": f"{coding_default} ↔ {p['coder']} (Eskalation)",
|
||||
"escalate_chars": p["coding_escalate_chars"],
|
||||
},
|
||||
],
|
||||
# Rückwärtskompatible Flach-Liste (alte UI/Clients).
|
||||
"routes": [
|
||||
{"name": "chat", "target": f"{p['fast']} ↔ {p['heavy']} (nach Komplexität)"},
|
||||
{"name": "coding", "target": f"{coding_default} ↔ {p['coder']} (Eskalation)"},
|
||||
{"name": "<alias>", "target": "llama-swap-Passthrough (lädt bei Bedarf)"},
|
||||
],
|
||||
"heavy_threshold_chars": p["heavy_chars"],
|
||||
"fallbacks": [],
|
||||
"context_window_fallbacks": [],
|
||||
}
|
||||
|
||||
|
||||
def gateway_reachable() -> bool:
|
||||
# Der eingebaute Gateway lebt in MC und proxyt llama-swap → erreichbar, wenn Engine läuft.
|
||||
return engine_reachable()
|
||||
@@ -0,0 +1,41 @@
|
||||
"""Token-Erfassung für den Builtin-Gateway.
|
||||
|
||||
Parst die `usage`-Felder aus llama-swap-Antworten (Stream + Non-Stream) und meldet
|
||||
sie an token_stats. Hält den gateway_proxy-Router dünn und ersetzt die zuvor inline
|
||||
verstreute, still scheiternde String-Suche durch einen testbaren SSE-Zeilenparser.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
|
||||
from services.token_stats import increment_tokens
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def record_usage(usage: dict | None, model: str) -> None:
|
||||
"""Ein usage-Objekt verbuchen (no-op bei None/leer)."""
|
||||
if not usage:
|
||||
return
|
||||
prompt = usage.get("prompt_tokens", 0)
|
||||
completion = usage.get("completion_tokens", 0)
|
||||
if prompt or completion:
|
||||
increment_tokens(prompt, completion, model=model)
|
||||
|
||||
|
||||
def record_stream_chunk(chunk: bytes, model: str) -> None:
|
||||
"""Rohen SSE-Chunk auf `usage` prüfen und Tokens verbuchen. Fehler werden
|
||||
geloggt (debug) statt verschluckt — ein defekter Chunk bricht den Stream nicht."""
|
||||
if b'"usage"' not in chunk:
|
||||
return
|
||||
text = chunk.decode("utf-8", errors="ignore")
|
||||
for line in text.splitlines():
|
||||
if not line.startswith("data:"):
|
||||
continue
|
||||
data_str = line[5:].strip()
|
||||
if not data_str or data_str == "[DONE]":
|
||||
continue
|
||||
try:
|
||||
record_usage(json.loads(data_str).get("usage"), model)
|
||||
except json.JSONDecodeError:
|
||||
log.debug("gateway stream: usage-Parsing fehlgeschlagen: %s", data_str[:120])
|
||||
@@ -0,0 +1,161 @@
|
||||
"""
|
||||
GGUF-Tokenizer-Fingerprint — liest die Tokenizer-Identität direkt aus dem
|
||||
GGUF-Header (ohne das Modell zu laden), um zu entscheiden, ob ein Draft-Modell
|
||||
**vocab-kompatibel** mit einem Ziel-Modell ist (Voraussetzung für Speculative
|
||||
Decoding in llama.cpp — sonst: "draft model vocab type must match target").
|
||||
|
||||
Wir lesen nur die Metadaten-KV-Sektion am Dateianfang und brechen ab, sobald
|
||||
`tokenizer.ggml.tokens` erreicht ist (dessen Länge = n_vocab). model+pre+n_vocab
|
||||
identifizieren den Tokenizer eindeutig genug, um die in der Praxis relevanten
|
||||
Fälle zu unterscheiden (Qwen2.5 vs Qwen3 vs Qwen3.6 etc.). Die llama.cpp-Prüfung
|
||||
beim Laden bleibt der letzte Schiedsrichter.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import struct
|
||||
from functools import lru_cache
|
||||
|
||||
# GGUF value types (https://github.com/ggml-org/ggml/blob/master/docs/gguf.md)
|
||||
_T_UINT8, _T_INT8, _T_UINT16, _T_INT16, _T_UINT32, _T_INT32, _T_FLOAT32, \
|
||||
_T_BOOL, _T_STRING, _T_ARRAY, _T_UINT64, _T_INT64, _T_FLOAT64 = range(13)
|
||||
|
||||
_SCALAR_FMT = {
|
||||
_T_UINT8: "<B", _T_INT8: "<b", _T_UINT16: "<H", _T_INT16: "<h",
|
||||
_T_UINT32: "<I", _T_INT32: "<i", _T_FLOAT32: "<f", _T_BOOL: "<?",
|
||||
_T_UINT64: "<Q", _T_INT64: "<q", _T_FLOAT64: "<d",
|
||||
}
|
||||
_SCALAR_SIZE = {t: struct.calcsize(f) for t, f in _SCALAR_FMT.items()}
|
||||
|
||||
_WANT_STRINGS = {"tokenizer.ggml.model", "tokenizer.ggml.pre", "general.architecture"}
|
||||
|
||||
|
||||
class _Reader:
|
||||
def __init__(self, f):
|
||||
self.f = f
|
||||
|
||||
def read(self, n: int) -> bytes:
|
||||
b = self.f.read(n)
|
||||
if len(b) != n:
|
||||
raise EOFError("unerwartetes Dateiende beim GGUF-Parsen")
|
||||
return b
|
||||
|
||||
def u32(self) -> int:
|
||||
return struct.unpack("<I", self.read(4))[0]
|
||||
|
||||
def u64(self) -> int:
|
||||
return struct.unpack("<Q", self.read(8))[0]
|
||||
|
||||
def gstr(self) -> str:
|
||||
n = self.u64()
|
||||
return self.read(n).decode("utf-8", "replace")
|
||||
|
||||
def skip_value(self, vtype: int) -> None:
|
||||
"""Liest einen Wert und verwirft ihn (um den Datei-Pointer korrekt
|
||||
weiterzuschieben). Arrays werden elementweise konsumiert."""
|
||||
if vtype == _T_STRING:
|
||||
self.f.seek(self.u64(), 1)
|
||||
elif vtype in _SCALAR_SIZE:
|
||||
self.f.seek(_SCALAR_SIZE[vtype], 1)
|
||||
elif vtype == _T_ARRAY:
|
||||
etype = self.u32()
|
||||
count = self.u64()
|
||||
if etype == _T_STRING:
|
||||
for _ in range(count):
|
||||
self.f.seek(self.u64(), 1)
|
||||
elif etype in _SCALAR_SIZE:
|
||||
self.f.seek(_SCALAR_SIZE[etype] * count, 1)
|
||||
else:
|
||||
raise ValueError(f"unbekannter Array-Elementtyp {etype}")
|
||||
else:
|
||||
raise ValueError(f"unbekannter GGUF-Wertetyp {vtype}")
|
||||
|
||||
|
||||
def _read_fingerprint(path: str) -> dict | None:
|
||||
"""Liest model/pre/n_vocab aus dem GGUF-Header. None bei Fehler/kein GGUF."""
|
||||
try:
|
||||
with open(path, "rb") as fh:
|
||||
r = _Reader(fh)
|
||||
if r.read(4) != b"GGUF":
|
||||
return None
|
||||
r.u32() # version
|
||||
r.u64() # tensor_count
|
||||
kv_count = r.u64()
|
||||
fp: dict = {"model": None, "pre": None, "arch": None, "n_vocab": None,
|
||||
"tokens_sha": None}
|
||||
for _ in range(kv_count):
|
||||
key = r.gstr()
|
||||
vtype = r.u32()
|
||||
if key == "tokenizer.ggml.tokens" and vtype == _T_ARRAY:
|
||||
etype = r.u32()
|
||||
count = r.u64()
|
||||
fp["n_vocab"] = count
|
||||
if etype != _T_STRING:
|
||||
return None
|
||||
# ECHTE Vocab-Identität: sha256 über die tatsächliche Token-Liste
|
||||
# (familienunabhängig — funktioniert für Qwen, Llama, Mistral, …).
|
||||
h = hashlib.sha256()
|
||||
h.update(count.to_bytes(8, "little"))
|
||||
for _ in range(count):
|
||||
n = r.u64()
|
||||
h.update(r.read(n))
|
||||
fp["tokens_sha"] = h.hexdigest()
|
||||
# model/pre kommen vor tokens → wir haben alles. Abbrechen.
|
||||
break
|
||||
if key in _WANT_STRINGS and vtype == _T_STRING:
|
||||
val = r.gstr()
|
||||
if key == "tokenizer.ggml.model":
|
||||
fp["model"] = val
|
||||
elif key == "tokenizer.ggml.pre":
|
||||
fp["pre"] = val
|
||||
else:
|
||||
fp["arch"] = val
|
||||
else:
|
||||
r.skip_value(vtype)
|
||||
if fp["model"] is None and fp["n_vocab"] is None:
|
||||
return None
|
||||
return fp
|
||||
except (OSError, EOFError, ValueError, struct.error):
|
||||
return None
|
||||
|
||||
|
||||
@lru_cache(maxsize=256)
|
||||
def _cached(path: str, mtime: float, size: int) -> tuple | None:
|
||||
fp = _read_fingerprint(path)
|
||||
if fp is None:
|
||||
return None
|
||||
return (fp.get("model"), fp.get("pre"), fp.get("n_vocab"), fp.get("arch"), fp.get("tokens_sha"))
|
||||
|
||||
|
||||
def fingerprint(path: str) -> dict | None:
|
||||
"""Tokenizer-Fingerprint eines GGUF (gecacht nach Pfad+mtime+size).
|
||||
Returns dict(model, pre, n_vocab, arch, tokens_sha) oder None wenn nicht lesbar."""
|
||||
import os
|
||||
try:
|
||||
st = os.stat(path)
|
||||
except OSError:
|
||||
return None
|
||||
t = _cached(path, st.st_mtime, st.st_size)
|
||||
if t is None:
|
||||
return None
|
||||
return {"model": t[0], "pre": t[1], "n_vocab": t[2], "arch": t[3], "tokens_sha": t[4]}
|
||||
|
||||
|
||||
def vocab_key(path: str) -> tuple | None:
|
||||
"""ECHTER Vergleichsschlüssel für Vocab-Kompatibilität: (model, pre, n_vocab, sha256
|
||||
der vollständigen Token-Liste). Vergleicht den TATSÄCHLICHEN Vokabular-Inhalt, nicht
|
||||
nur Metadaten — familienunabhängig (Qwen, Llama, Mistral, …). Genau diese Identität
|
||||
verlangt llama.cpp für Speculative Decoding."""
|
||||
fp = fingerprint(path)
|
||||
if not fp or fp["n_vocab"] is None or not fp.get("tokens_sha"):
|
||||
return None
|
||||
return (fp["model"], fp["pre"], fp["n_vocab"], fp["tokens_sha"])
|
||||
|
||||
|
||||
def compatible(target_path: str, draft_path: str) -> bool | None:
|
||||
"""True/False ob draft vocab-kompatibel zum target ist. None = unbestimmbar
|
||||
(eine Datei nicht lesbar) → UI behandelt das als 'nicht bestätigt'."""
|
||||
a = vocab_key(target_path)
|
||||
b = vocab_key(draft_path)
|
||||
if a is None or b is None:
|
||||
return None
|
||||
return a == b
|
||||
@@ -0,0 +1,103 @@
|
||||
"""HuggingFace-Helfer: GGUF-Dateien eines Repos auflösen (inkl. Split-Teile) + Größen,
|
||||
freie Suche, Repo-URL→ID, verfügbare Quants."""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
import httpx
|
||||
|
||||
|
||||
def normalize_repo(s: str) -> str:
|
||||
"""Akzeptiert volle HF-URL oder `org/repo` → liefert immer `org/repo`."""
|
||||
s = (s or "").strip()
|
||||
m = re.search(r"huggingface\.co/([^/\s]+/[^/\s?#]+)", s)
|
||||
if m:
|
||||
return m.group(1)
|
||||
return s.strip("/")
|
||||
|
||||
|
||||
def list_quants(repo: str) -> list[str]:
|
||||
"""Verfügbare Quant-Stufen eines Repos (aus den GGUF-Dateinamen, ohne mmproj)."""
|
||||
quants: set[str] = set()
|
||||
for e in _tree(repo):
|
||||
p = str(e.get("path", ""))
|
||||
if p.lower().endswith(".gguf") and "mmproj" not in p.lower():
|
||||
m = re.search(r"(I?Q\d[\w]*|F16|BF16|FP16|F32)", p, re.IGNORECASE)
|
||||
if m:
|
||||
quants.add(m.group(1).upper())
|
||||
# gängige Reihenfolge zuerst
|
||||
order = {"Q4_K_M": 0, "Q4_K_S": 1, "Q5_K_M": 2, "Q6_K": 3, "Q8_0": 4, "Q3_K_M": 5, "Q2_K": 6}
|
||||
return sorted(quants, key=lambda q: (order.get(q, 99), q))
|
||||
|
||||
|
||||
def search(q: str = "", limit: int = 24) -> list[dict]:
|
||||
"""Freie HF-Suche nach GGUF-Repos. Ohne q → Top-GGUF nach Downloads (Stöbern)."""
|
||||
url = (f"https://huggingface.co/api/models?filter=gguf"
|
||||
f"&sort=downloads&direction=-1&limit={limit}")
|
||||
if q and q.strip():
|
||||
url += f"&search={q.strip()}"
|
||||
try:
|
||||
with httpx.Client(timeout=12.0) as c:
|
||||
data = c.get(url).json()
|
||||
except Exception:
|
||||
return []
|
||||
out = []
|
||||
for m in (data if isinstance(data, list) else []):
|
||||
rid = m.get("id")
|
||||
if rid:
|
||||
out.append({"repo": rid, "downloads": int(m.get("downloads") or 0),
|
||||
"likes": int(m.get("likes") or 0)})
|
||||
return out
|
||||
|
||||
|
||||
def hf_bin() -> str:
|
||||
"""Pfad zur `hf`-CLI (bevorzugt neben dem laufenden Python im venv)."""
|
||||
cand = os.path.join(os.path.dirname(sys.executable), "hf")
|
||||
return cand if os.path.exists(cand) else "hf"
|
||||
|
||||
|
||||
def _tree(repo: str) -> list[dict]:
|
||||
url = f"https://huggingface.co/api/models/{repo}/tree/main?recursive=true"
|
||||
with httpx.Client(timeout=20.0) as c:
|
||||
data = c.get(url).json()
|
||||
return data if isinstance(data, list) else []
|
||||
|
||||
|
||||
def _size(entry: dict) -> int:
|
||||
return int(entry.get("size") or (entry.get("lfs") or {}).get("size") or 0)
|
||||
|
||||
|
||||
def resolve_gguf(repo: str, quant: str = "Q4_K_M") -> dict:
|
||||
"""Beste GGUF-Auswahl eines Repos für einen Quant. Behandelt Split-GGUFs
|
||||
(-00001-of-000NN) als Gruppe. Liefert die Datei-/Pattern-Infos für den Download.
|
||||
|
||||
Rückgabe: {files:[paths], first:path, total_bytes:int, mmproj:path|None, split:bool}
|
||||
"""
|
||||
tree = _tree(repo)
|
||||
ggufs = [e for e in tree if str(e.get("path", "")).lower().endswith(".gguf")]
|
||||
q = quant.lower()
|
||||
# mmproj separat (Vision-Projektor)
|
||||
mmproj = next((e["path"] for e in ggufs if "mmproj" in e["path"].lower()), None)
|
||||
model = [e for e in ggufs if "mmproj" not in e["path"].lower()]
|
||||
# bevorzugt den gewünschten Quant
|
||||
pref = [e for e in model if q in e["path"].lower()]
|
||||
chosen = pref or model
|
||||
if not chosen:
|
||||
return {"files": [], "first": None, "total_bytes": 0, "mmproj": mmproj, "split": False}
|
||||
# Split? Wenn die gewählten Dateien -of- enthalten → alle Teile dieser Gruppe.
|
||||
split = any("-of-" in e["path"].lower() for e in chosen)
|
||||
if split:
|
||||
parts = sorted([e for e in chosen if "-of-" in e["path"].lower()], key=lambda e: e["path"])
|
||||
files = [e["path"] for e in parts]
|
||||
first = files[0]
|
||||
total = sum(_size(e) for e in parts)
|
||||
else:
|
||||
# ein einzelnes File: nimm das kleinste passende (typisch genau eins)
|
||||
chosen.sort(key=lambda e: _size(e))
|
||||
first = chosen[0]["path"]
|
||||
files = [first]
|
||||
total = _size(chosen[0])
|
||||
if mmproj:
|
||||
total += next((_size(e) for e in ggufs if e["path"] == mmproj), 0)
|
||||
return {"files": files, "first": first, "total_bytes": total, "mmproj": mmproj, "split": split}
|
||||
@@ -0,0 +1,195 @@
|
||||
"""
|
||||
Mini-Job-System: Hintergrund-Prozesse mit Live-Log + Download-Fortschritt.
|
||||
Portiert aus Mission Control v1 (jobengine.py). In-Memory, ein Daemon-Thread je Job.
|
||||
"""
|
||||
|
||||
import glob
|
||||
import os
|
||||
import shlex
|
||||
import subprocess
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
|
||||
JOBS: dict[str, dict] = {}
|
||||
_PROCS: dict[str, subprocess.Popen] = {}
|
||||
_LOG_CAP = 400
|
||||
|
||||
|
||||
def _append_log(job: dict, line: str) -> None:
|
||||
job["log"].append(line)
|
||||
if len(job["log"]) > _LOG_CAP:
|
||||
del job["log"][0]
|
||||
|
||||
|
||||
def _pump_output(job: dict, stream) -> None:
|
||||
"""Liest byteweise; `\\r` (tqdm/hf-Fortschritt) überschreibt die letzte Zeile."""
|
||||
buf = b""
|
||||
overwrite = False
|
||||
pending_cr = False
|
||||
|
||||
def commit():
|
||||
line = buf.decode("utf-8", "replace")
|
||||
if overwrite and job["log"]:
|
||||
job["log"][-1] = line
|
||||
else:
|
||||
_append_log(job, line)
|
||||
|
||||
while True:
|
||||
ch = stream.read(1)
|
||||
if not ch:
|
||||
break
|
||||
if pending_cr:
|
||||
pending_cr = False
|
||||
if ch == b"\n":
|
||||
commit(); overwrite = False; buf = b""
|
||||
continue
|
||||
commit(); overwrite = True; buf = b""
|
||||
if ch == b"\r":
|
||||
pending_cr = True
|
||||
elif ch == b"\n":
|
||||
commit(); overwrite = False; buf = b""
|
||||
else:
|
||||
buf += ch
|
||||
if pending_cr:
|
||||
commit(); overwrite = True; buf = b""
|
||||
if buf:
|
||||
commit()
|
||||
|
||||
|
||||
def _run_job(job_id: str, args: list[str], env: dict | None = None, sudo_password: str | None = None):
|
||||
job = JOBS[job_id]
|
||||
job["state"] = "running"
|
||||
try:
|
||||
actual_args = list(args)
|
||||
if sudo_password is not None:
|
||||
for i, arg in enumerate(actual_args):
|
||||
if isinstance(arg, str):
|
||||
actual_args[i] = arg.replace("sudo -n", "sudo -S").replace("sudo ", "sudo -S ")
|
||||
|
||||
proc = subprocess.Popen(
|
||||
actual_args, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
stdin=subprocess.PIPE if sudo_password is not None else None,
|
||||
bufsize=0,
|
||||
env={**os.environ, **(env or {})},
|
||||
)
|
||||
_PROCS[job_id] = proc
|
||||
|
||||
if sudo_password is not None and proc.stdin:
|
||||
proc.stdin.write((sudo_password + "\n").encode("utf-8"))
|
||||
proc.stdin.flush()
|
||||
proc.stdin.close()
|
||||
|
||||
_pump_output(job, proc.stdout)
|
||||
proc.wait()
|
||||
job["returncode"] = proc.returncode
|
||||
job["state"] = "canceled" if job.get("canceled") else ("done" if proc.returncode == 0 else "failed")
|
||||
|
||||
# Check if failed due to sudo authorization failure
|
||||
if proc.returncode != 0 and job["log"]:
|
||||
log_str = "\n".join(job["log"])
|
||||
if "a password is required" in log_str or "password" in log_str.lower() or "sudo:" in log_str:
|
||||
job["sudo_failed"] = True
|
||||
except Exception as exc: # noqa: BLE001
|
||||
_append_log(job, f"[mc] Fehler: {exc}")
|
||||
job["state"] = "failed"
|
||||
job["returncode"] = -1
|
||||
finally:
|
||||
_PROCS.pop(job_id, None)
|
||||
job["finished_at"] = time.time()
|
||||
cb = job.pop("_on_done", None)
|
||||
if cb and job["state"] == "done":
|
||||
try:
|
||||
cb()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
_append_log(job, f"[mc] Nachbearbeitung-Fehler: {exc}")
|
||||
|
||||
|
||||
def attach_download_progress(job_id: str, local_dir: str, total_bytes: int) -> None:
|
||||
"""Fortschritt in % aus wachsenden *.incomplete-Dateien (hf schreibt sie)."""
|
||||
if not total_bytes or total_bytes <= 0:
|
||||
return
|
||||
job = JOBS.get(job_id)
|
||||
if job is not None:
|
||||
job["progress"] = 0
|
||||
job["total_bytes"] = total_bytes
|
||||
|
||||
def _watch():
|
||||
pat = os.path.join(local_dir, ".cache", "huggingface", "download", "**", "*.incomplete")
|
||||
prev_t = prev_b = None
|
||||
rate = 0.0
|
||||
while True:
|
||||
j = JOBS.get(job_id)
|
||||
if not j or j["state"] in ("done", "failed", "canceled"):
|
||||
break
|
||||
try:
|
||||
inc = glob.glob(pat, recursive=True)
|
||||
cur = sum(os.path.getsize(f) for f in inc) if inc else 0
|
||||
if cur:
|
||||
j["progress"] = min(99, int(cur * 100 / total_bytes))
|
||||
j["done_bytes"] = cur
|
||||
now = time.time()
|
||||
if prev_t is not None and now > prev_t and cur >= prev_b:
|
||||
inst = (cur - prev_b) / (now - prev_t)
|
||||
rate = inst if rate == 0 else 0.3 * inst + 0.7 * rate
|
||||
if rate > 0:
|
||||
j["rate_bps"] = rate
|
||||
j["eta_s"] = int((total_bytes - cur) / rate)
|
||||
prev_t, prev_b = now, cur
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
time.sleep(1.0)
|
||||
j = JOBS.get(job_id)
|
||||
if j and j["state"] == "done":
|
||||
j["progress"] = 100
|
||||
j.pop("eta_s", None)
|
||||
|
||||
threading.Thread(target=_watch, daemon=True).start()
|
||||
|
||||
|
||||
def start_job(args: list[str], label: str, env: dict | None = None, on_done=None,
|
||||
sudo_password: str | None = None, group: str | None = None) -> str:
|
||||
job_id = uuid.uuid4().hex[:12]
|
||||
# Mask password in log if present in args
|
||||
log_args = list(args)
|
||||
JOBS[job_id] = {
|
||||
"id": job_id, "label": label, "state": "queued", "group": group,
|
||||
"log": ["$ " + " ".join(shlex.quote(a) for a in log_args)],
|
||||
"returncode": None, "started_at": time.time(), "finished_at": None,
|
||||
}
|
||||
if on_done:
|
||||
JOBS[job_id]["_on_done"] = on_done
|
||||
threading.Thread(target=_run_job, args=(job_id, args, env, sudo_password), daemon=True).start()
|
||||
return job_id
|
||||
|
||||
|
||||
def active_in_group(group: str) -> dict | None:
|
||||
"""Erster laufender/wartender Job einer Gruppe (z.B. 'maintenance'), sonst None.
|
||||
Basis für den Wartungs-Riegel: nur EIN System-Update gleichzeitig."""
|
||||
for j in JOBS.values():
|
||||
if j.get("group") == group and j.get("state") in ("running", "queued"):
|
||||
return j
|
||||
return None
|
||||
|
||||
|
||||
def cancel_job(job_id: str) -> bool:
|
||||
job = JOBS.get(job_id)
|
||||
if not job or job["state"] in ("done", "failed", "canceled"):
|
||||
return False
|
||||
job["canceled"] = True
|
||||
_append_log(job, "[mc] Abbruch angefordert…")
|
||||
proc = _PROCS.get(job_id)
|
||||
if proc is not None:
|
||||
try:
|
||||
proc.terminate()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
else:
|
||||
job["state"] = "canceled"
|
||||
job["finished_at"] = time.time()
|
||||
return True
|
||||
|
||||
|
||||
def public_jobs() -> list[dict]:
|
||||
"""Jobs ohne interne Felder (_on_done) für die API."""
|
||||
return [{k: v for k, v in j.items() if not k.startswith("_")} for j in JOBS.values()]
|
||||
@@ -0,0 +1,552 @@
|
||||
"""
|
||||
Engine-Service: liest/schreibt die llama-swap config.yaml und spricht die
|
||||
llama-swap-API. Portiert & erweitert aus Mission Control v1.
|
||||
|
||||
NEU in 2.0: `groups` für Ko-Residenz (schnell + schwer gleichzeitig geladen,
|
||||
`swap:false`) → Multi-Model-Delegation ohne Nachlade-Latenz.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
|
||||
import httpx
|
||||
from ruamel.yaml.scalarstring import LiteralScalarString
|
||||
|
||||
from config import (
|
||||
CMD_TEMPLATE, CONFIG_PATH, DEFAULT_TTL, DRAFTS_DIR, LLAMA_SWAP_URL,
|
||||
SPEC_DRAFT_MODEL_PATH, SPEC_DRAFT_N_MAX, SPEC_TYPE,
|
||||
)
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Kanonische Serving-Rollen — EINE Quelle der Wahrheit (identisch zu sources.ROLE_IDS,
|
||||
# maintenance, frontend ModelBadges.ROLES). `hermes` = Lucys Agent-Hirn (warm + ko-resident
|
||||
# in der `brains`-Gruppe); UI-Label „Hirn".
|
||||
ROLE_IDS = {"fast", "heavy", "coder", "vision", "scout", "hermes"}
|
||||
|
||||
_CTX_RE = re.compile(r"-(?:c|-ctx-size)\s+(\d+)")
|
||||
_PATH_RE = re.compile(r"-(?:m|-model)\s+([^\s]+)")
|
||||
_QUANT_RE = re.compile(r"(Q\d_[A-Z0-9_]+|IQ\d_[A-Z0-9_]+|fp16|bf16)\.gguf", re.IGNORECASE)
|
||||
_SPLIT_RE = re.compile(r"-(\d+)-of-(\d+)\.gguf$", re.IGNORECASE)
|
||||
|
||||
|
||||
def _gguf_total_size(path: str) -> int | None:
|
||||
"""Gesamtgröße eines GGUF inkl. ALLER Split-Teile (…-00001-of-00003.gguf).
|
||||
Die Größe nur des ersten Teils ist bei Splits irreführend (oft nur ein Header)."""
|
||||
try:
|
||||
base = os.path.basename(path)
|
||||
m = _SPLIT_RE.search(base)
|
||||
if not m:
|
||||
return os.path.getsize(path)
|
||||
prefix, dirn = base[:m.start()], os.path.dirname(path)
|
||||
total = sum(os.path.getsize(os.path.join(dirn, f))
|
||||
for f in os.listdir(dirn)
|
||||
if f.startswith(prefix) and _SPLIT_RE.search(f))
|
||||
return total or os.path.getsize(path)
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
|
||||
# --- Lesen -------------------------------------------------------------------
|
||||
def read_config() -> dict:
|
||||
if not CONFIG_PATH.exists():
|
||||
return {"models": {}}
|
||||
from ruamel.yaml import YAML
|
||||
r_yaml = YAML()
|
||||
r_yaml.preserve_quotes = True
|
||||
with CONFIG_PATH.open("r", encoding="utf-8") as f:
|
||||
data = r_yaml.load(f) or {}
|
||||
if not data.get("models"):
|
||||
data["models"] = {}
|
||||
return data
|
||||
|
||||
|
||||
def _parse_model(name: str, spec: dict) -> dict:
|
||||
spec = spec or {}
|
||||
cmd = str(spec.get("cmd", "")).strip()
|
||||
ctx = int(m.group(1)) if (m := _CTX_RE.search(cmd)) else None
|
||||
|
||||
path = filename = quant = ""
|
||||
size_bytes = None
|
||||
if (m := _PATH_RE.search(cmd)):
|
||||
path = m.group(1).replace("'", "").replace('"', "")
|
||||
filename = os.path.basename(path)
|
||||
if os.path.exists(path):
|
||||
size_bytes = _gguf_total_size(path)
|
||||
if (q := _QUANT_RE.search(path)):
|
||||
quant = q.group(1).upper()
|
||||
|
||||
aliases = spec.get("aliases") or []
|
||||
if isinstance(aliases, str):
|
||||
aliases = [aliases]
|
||||
aliases = [str(a) for a in aliases]
|
||||
role = aliases[0].lower() if aliases else (name.lower() if name.lower() in ROLE_IDS else None)
|
||||
|
||||
prompt_cache = "--prompt-cache " in cmd or cmd.endswith("--prompt-cache") or "--prompt-cache-all" in cmd
|
||||
# Draft-Modell erkennen — klassisch (--spec-draft-model) ODER MTP (--model-draft / -md).
|
||||
spec_draft = None
|
||||
if (m_draft := re.search(r"--(?:spec-draft-model|model-draft)\s+([^\s]+)", cmd)) \
|
||||
or (m_draft := re.search(r"(?<![\w-])-md\s+([^\s]+)", cmd)):
|
||||
spec_draft = os.path.basename(m_draft.group(1).replace("'", "").replace('"', ""))
|
||||
spec_type = None
|
||||
if (m_st := re.search(r"--spec-type\s+([^\s]+)", cmd)):
|
||||
spec_type = m_st.group(1)
|
||||
# Spec ist nur AKTIV, wenn BEIDES gesetzt ist (Draft-Modell UND --spec-type).
|
||||
spec_active = bool(spec_draft and spec_type)
|
||||
parallel_match = re.search(r"--parallel\s+(\d+)", cmd)
|
||||
parallel_slots = int(parallel_match.group(1)) if parallel_match else 1
|
||||
|
||||
from services.caps import capabilities
|
||||
return {
|
||||
"name": name,
|
||||
"role": role,
|
||||
"aliases": aliases,
|
||||
"api_ids": [name] + aliases,
|
||||
"ctx": ctx,
|
||||
"ttl": spec.get("ttl"),
|
||||
"cmd": cmd,
|
||||
"gguf_path": path,
|
||||
"filename": filename,
|
||||
"quant": quant,
|
||||
"size_bytes": size_bytes,
|
||||
"incomplete": not path,
|
||||
"prompt_cache": prompt_cache,
|
||||
"spec_draft_model": spec_draft,
|
||||
"spec_type": spec_type,
|
||||
"spec_active": spec_active,
|
||||
"parallel_slots": parallel_slots,
|
||||
"capabilities": capabilities(
|
||||
name=filename or name, cmd=cmd,
|
||||
gguf_path=(path if (path and os.path.exists(path)) else ""),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def list_models() -> list[dict]:
|
||||
cfg = read_config()
|
||||
return [_parse_model(name, spec) for name, spec in (cfg.get("models") or {}).items()]
|
||||
|
||||
|
||||
def engine_reachable() -> bool:
|
||||
try:
|
||||
with httpx.Client(timeout=3.0) as c:
|
||||
return c.get(f"{LLAMA_SWAP_URL}/v1/models").status_code == 200
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
# --- Schreiben ---------------------------------------------------------------
|
||||
def model_id_from_path(model_path: str) -> str:
|
||||
"""Sprechende Modell-ID (= API-Name) aus dem GGUF-Pfad: Repo-Ordnername ohne
|
||||
'-GGUF'. Fallback: Dateiname ohne Quant-Suffix.
|
||||
Split-GGUFs liegen oft in einem Quant-Unterordner (…/Q4_K_M/file-00001-of-…) →
|
||||
dann eine Ebene höher (Repo-Ordner) nehmen, sonst hieße das Modell 'Q4_K_M'."""
|
||||
d = os.path.basename(os.path.dirname(model_path))
|
||||
if re.fullmatch(r"(I?Q\d[\w]*|UD-Q\d[\w]*|F16|BF16|FP16|F32)", d, flags=re.I):
|
||||
d = os.path.basename(os.path.dirname(os.path.dirname(model_path)))
|
||||
name = re.sub(r"[-_]?GGUF$", "", d, flags=re.I).strip("-_")
|
||||
if not name:
|
||||
fn = re.sub(r"\.gguf$", "", os.path.basename(model_path), flags=re.I)
|
||||
fn = re.sub(r"-\d+-of-\d+$", "", fn)
|
||||
name = re.sub(r"[-_](Q\d[\w]*|IQ\d[\w]*|F16|BF16|FP16|F32)$", "", fn, flags=re.I)
|
||||
return name or "modell"
|
||||
|
||||
|
||||
def set_role_alias(cfg: dict, model_id: str, role: str | None) -> None:
|
||||
"""Rolle als eindeutigen llama-swap-`aliases`-Eintrag setzen (vorher bei allen
|
||||
anderen Modellen entfernen). role=None/leer entfernt den Alias."""
|
||||
models = cfg.get("models") or {}
|
||||
role = (role or "").strip().lower()
|
||||
if role:
|
||||
for mid, spec in models.items():
|
||||
if mid == model_id or not isinstance(spec, dict):
|
||||
continue
|
||||
al = [a for a in (spec.get("aliases") or []) if str(a).lower() != role]
|
||||
if al:
|
||||
spec["aliases"] = al
|
||||
else:
|
||||
spec.pop("aliases", None)
|
||||
spec = models.get(model_id)
|
||||
if isinstance(spec, dict):
|
||||
if role and role != model_id.lower():
|
||||
spec["aliases"] = [role]
|
||||
else:
|
||||
spec.pop("aliases", None)
|
||||
|
||||
|
||||
def _augment_vision(cmd: str, model_path: str, mmproj_path: str | None) -> str:
|
||||
"""Vision-Modelle brauchen --mmproj <projektor> und --jinja."""
|
||||
if mmproj_path:
|
||||
if "--mmproj" not in cmd:
|
||||
cmd += f" --mmproj {mmproj_path}"
|
||||
if "--jinja" not in cmd:
|
||||
cmd += " --jinja"
|
||||
return cmd
|
||||
|
||||
|
||||
def write_config(cfg: dict) -> None:
|
||||
"""Atomar schreiben (tmp + os.replace), damit llama-swap mit -watch-config nie
|
||||
eine halbe Datei sieht. Fehlende Schreibrechte → klare Meldung."""
|
||||
try:
|
||||
CONFIG_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = CONFIG_PATH.with_name(CONFIG_PATH.name + ".tmp")
|
||||
from ruamel.yaml import YAML
|
||||
r_yaml = YAML()
|
||||
r_yaml.preserve_quotes = True
|
||||
with tmp.open("w", encoding="utf-8") as f:
|
||||
r_yaml.dump(cfg, f)
|
||||
os.replace(tmp, CONFIG_PATH)
|
||||
except PermissionError as exc:
|
||||
raise PermissionError(
|
||||
f"Mission Control darf '{CONFIG_PATH}' nicht schreiben. "
|
||||
f"Einmalig: sudo chown -R hitonabi:hitonabi {CONFIG_PATH.parent}"
|
||||
) from exc
|
||||
|
||||
|
||||
|
||||
def register_model(model_path: str, role: str | None = None, ctx: int = 8192,
|
||||
ttl: int | None = None, mmproj_path: str | None = None,
|
||||
jinja: bool = False) -> str:
|
||||
"""Ein GGUF als llama-swap-Modell eintragen (cmd + Rolle-Alias). Gibt die
|
||||
Modell-ID zurück. jinja=True erzwingt --jinja (Tool-Calling, z.B. fürs Agent-Hirn)."""
|
||||
cfg = read_config()
|
||||
model_id = model_id_from_path(model_path)
|
||||
cmd = CMD_TEMPLATE.replace("{model}", model_path).replace("{ctx}", str(ctx))
|
||||
cmd = _augment_vision(cmd, model_path, mmproj_path)
|
||||
if jinja and "--jinja" not in cmd:
|
||||
cmd += " --jinja"
|
||||
|
||||
role_lower = (role or "").strip().lower()
|
||||
# KV-Cache-Reuse über Turns (Prompt-Cache wiederverwenden) — hilft allen Chat-Modellen
|
||||
# (Agent-Hirn, Coding, Multi-Turn). Spiegelt die auf der Box bewährten Flags wider, damit
|
||||
# neu installierte Modelle nicht hinter dem hand-getunten Stand zurückbleiben (Drift-Fix).
|
||||
# NICHT bei Vision-Modellen: --cache-reuse + --mmproj bricht llama-server (live verifiziert,
|
||||
# deshalb fahren vision/scout auf der Box ohne cache-reuse).
|
||||
if "--cache-reuse" not in cmd and "--mmproj" not in cmd:
|
||||
cmd += " --cache-reuse 256 -cram 16384"
|
||||
# IDE-Coding profitiert von Nebenläufigkeit; sonst Default 1 Slot = voller Kontext/Anfrage
|
||||
# (--parallel teilt den Kontext HART auf die Slots auf, s. docs/OPTIMIZATION_PLAN.md §9.4 V6).
|
||||
if role_lower == "coder" and "--parallel" not in cmd:
|
||||
cmd += " --parallel 2"
|
||||
# Vocab-kompatiblen Draft automatisch anhängen — klassisch (DRAFTS_DIR) ODER MTP-Kopf neben
|
||||
# dem Modell. Self-guarding: ohne kompatiblen/vorhandenen Draft passiert nichts (später im UI
|
||||
# setzbar). Bei frischem Install existiert die Modell-GGUF noch nicht → ebenfalls kein Draft.
|
||||
if "--spec-draft-model" not in cmd and "--model-draft" not in cmd:
|
||||
cmd += spec_draft_flags(model_path)
|
||||
|
||||
cfg.setdefault("models", {})[model_id] = {
|
||||
"cmd": LiteralScalarString(cmd + "\n"),
|
||||
"ttl": ttl if ttl is not None else DEFAULT_TTL,
|
||||
}
|
||||
set_role_alias(cfg, model_id, role)
|
||||
write_config(cfg)
|
||||
return model_id
|
||||
|
||||
|
||||
# --- Speculative-Draft / Vocab-Kompatibilität --------------------------------
|
||||
def list_drafts() -> list[dict]:
|
||||
"""Alle Draft-GGUFs in DRAFTS_DIR mit Tokenizer-Fingerprint."""
|
||||
from services import gguf_meta
|
||||
out = []
|
||||
if DRAFTS_DIR.is_dir():
|
||||
for p in sorted(DRAFTS_DIR.glob("*.gguf")):
|
||||
out.append({
|
||||
"path": str(p), "filename": p.name,
|
||||
"size_bytes": p.stat().st_size if p.exists() else None,
|
||||
"vocab": gguf_meta.fingerprint(str(p)),
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def _is_mtp_draft(draft_path: str) -> bool:
|
||||
"""Ist dieser Draft ein MTP-Kopf (Multi-Token-Prediction) statt eines klassischen
|
||||
Draft-Modells? MTP-Köpfe (z.B. gemma-4) laden mit `--model-draft … --spec-type
|
||||
draft-mtp` statt `--spec-draft-model … --spec-type draft-simple`. Erkennung am Arch
|
||||
('…-assistant' / 'mtp') oder Dateinamen ('mtp-*', '*-MTP', '*-assistant')."""
|
||||
base = os.path.basename(draft_path).lower()
|
||||
if base.startswith("mtp-") or "-mtp" in base or "assistant" in base:
|
||||
return True
|
||||
from services import gguf_meta
|
||||
arch = ((gguf_meta.fingerprint(draft_path) or {}).get("arch") or "").lower()
|
||||
return arch.endswith("-assistant") or "mtp" in arch
|
||||
|
||||
|
||||
def _spec_flags_for_draft(draft_path: str) -> str:
|
||||
"""Korrekte llama-server-Spec-Flags für einen (vocab-kompatiblen) Draft. MTP-Kopf →
|
||||
`--model-draft … --spec-type draft-mtp --spec-draft-n-max N`; klassischer Draft →
|
||||
`--spec-draft-model … --spec-type draft-simple`. (Beides nötig, sonst Spec inaktiv.)"""
|
||||
if _is_mtp_draft(draft_path):
|
||||
return (f" --model-draft {draft_path} --spec-type draft-mtp"
|
||||
f" --spec-draft-n-max {SPEC_DRAFT_N_MAX}")
|
||||
return f" --spec-draft-model {draft_path} --spec-type {SPEC_TYPE}"
|
||||
|
||||
|
||||
def _sibling_mtp_drafters(target_path: str) -> list[str]:
|
||||
"""MTP-Kopf-GGUFs NEBEN dem Zielmodell (gleicher Ordner): 'mtp-*.gguf', '*-MTP.gguf',
|
||||
'*-assistant*.gguf'. Per Konstruktion vocab-identisch zum Modell → idealer Draft."""
|
||||
out: list[str] = []
|
||||
d = os.path.dirname(target_path)
|
||||
if os.path.isdir(d):
|
||||
for f in sorted(os.listdir(d)):
|
||||
fl = f.lower()
|
||||
if fl.endswith(".gguf") and (fl.startswith("mtp-") or "-mtp" in fl or "assistant" in fl):
|
||||
p = os.path.join(d, f)
|
||||
if p != target_path:
|
||||
out.append(p)
|
||||
return out
|
||||
|
||||
|
||||
def find_compatible_draft(target_path: str) -> str | None:
|
||||
"""Pfad eines vocab-kompatiblen Drafts für target_path, oder None.
|
||||
Bevorzugt einen MTP-Kopf NEBEN dem Modell (höchste Qualität, by-construction),
|
||||
dann MC_SPEC_DRAFT_MODEL (falls gesetzt+kompatibel), sonst der erste kompatible
|
||||
Draft in DRAFTS_DIR. None auch, wenn target (noch) fehlt (nicht verifizierbar →
|
||||
bewusst KEIN Draft anhängen)."""
|
||||
if not target_path or not os.path.exists(target_path):
|
||||
return None
|
||||
from services import gguf_meta
|
||||
candidates: list[str] = list(_sibling_mtp_drafters(target_path))
|
||||
if SPEC_DRAFT_MODEL_PATH and os.path.exists(SPEC_DRAFT_MODEL_PATH):
|
||||
candidates.append(SPEC_DRAFT_MODEL_PATH)
|
||||
for d in list_drafts():
|
||||
if d["path"] not in candidates:
|
||||
candidates.append(d["path"])
|
||||
for c in candidates:
|
||||
if gguf_meta.compatible(target_path, c) is True:
|
||||
return c
|
||||
return None
|
||||
|
||||
|
||||
def spec_draft_flags(target_path: str) -> str:
|
||||
"""llama-server-Flags für Speculative Decoding (Draft + --spec-type), oder ''
|
||||
wenn kein kompatibler Draft existiert. MTP-bewusst (s. _spec_flags_for_draft)."""
|
||||
d = find_compatible_draft(target_path)
|
||||
return _spec_flags_for_draft(d) if d else ""
|
||||
|
||||
|
||||
def drafts_for(target_path: str) -> dict:
|
||||
"""Für die UI: alle Drafts + ihre Kompatibilität zum Ziel-Modell. Schließt MTP-Köpfe
|
||||
NEBEN dem Zielmodell ein (DRAFTS_DIR kennt sie nicht). `mtp:true` markiert MTP-Drafts.
|
||||
compatible=None heißt 'nicht prüfbar' (Ziel- oder Draft-GGUF fehlt)."""
|
||||
from services import gguf_meta
|
||||
exists = bool(target_path and os.path.exists(target_path))
|
||||
drafts = list_drafts()
|
||||
seen = {d["path"] for d in drafts}
|
||||
for p in _sibling_mtp_drafters(target_path):
|
||||
if p not in seen:
|
||||
drafts.append({"path": p, "filename": os.path.basename(p),
|
||||
"size_bytes": os.path.getsize(p) if os.path.exists(p) else None,
|
||||
"vocab": gguf_meta.fingerprint(p)})
|
||||
for d in drafts:
|
||||
d["compatible"] = gguf_meta.compatible(target_path, d["path"]) if exists else None
|
||||
d["mtp"] = _is_mtp_draft(d["path"])
|
||||
return {
|
||||
"target_path": target_path,
|
||||
"target_exists": exists,
|
||||
"target_vocab": gguf_meta.fingerprint(target_path) if exists else None,
|
||||
"drafts": drafts,
|
||||
}
|
||||
|
||||
|
||||
def set_spec_draft(model_id: str, draft_path: str | None) -> dict:
|
||||
"""Setzt (oder entfernt mit draft_path=None) den Spec-Draft eines Modells.
|
||||
Validiert die Vocab-Kompatibilität — ein inkompatibler/unprüfbarer Draft wird
|
||||
abgelehnt (idiotensicher). Returns {ok, reason}."""
|
||||
cfg = read_config()
|
||||
spec = (cfg.get("models") or {}).get(model_id)
|
||||
if not isinstance(spec, dict):
|
||||
return {"ok": False, "reason": "Modell nicht gefunden"}
|
||||
cmd = str(spec.get("cmd", ""))
|
||||
# vorhandene Spec-Flags entfernen (idempotent) — klassisch UND MTP.
|
||||
cmd = re.sub(r"\s+--(?:spec-draft-model|model-draft)\s+\S+", "", cmd)
|
||||
cmd = re.sub(r"\s+-md\s+\S+", "", cmd)
|
||||
cmd = re.sub(r"\s+--spec-type\s+\S+", "", cmd)
|
||||
cmd = re.sub(r"\s+--spec-draft-n-(?:max|min)\s+\S+", "", cmd)
|
||||
|
||||
if draft_path:
|
||||
# relative Angabe (nur Dateiname) gegen DRAFTS_DIR auflösen
|
||||
if not os.path.isabs(draft_path) and "/" not in draft_path:
|
||||
draft_path = str(DRAFTS_DIR / draft_path)
|
||||
if not os.path.exists(draft_path):
|
||||
return {"ok": False, "reason": "Draft-Datei nicht gefunden"}
|
||||
from services import gguf_meta
|
||||
target = ""
|
||||
if (mt := _PATH_RE.search(cmd)):
|
||||
target = mt.group(1).replace("'", "").replace('"', "")
|
||||
comp = gguf_meta.compatible(target, draft_path) if os.path.exists(target) else None
|
||||
if comp is not True:
|
||||
reason = ("Draft ist NICHT vocab-kompatibel zum Modell — Speculative Decoding "
|
||||
"würde beim Laden scheitern."
|
||||
if comp is False else
|
||||
"Kompatibilität nicht prüfbar (Modell-GGUF fehlt) — Draft nicht gesetzt.")
|
||||
return {"ok": False, "reason": reason}
|
||||
cmd = cmd.rstrip() + _spec_flags_for_draft(draft_path)
|
||||
|
||||
spec["cmd"] = LiteralScalarString(cmd.rstrip() + "\n")
|
||||
write_config(cfg)
|
||||
return {"ok": True, "reason": ""}
|
||||
|
||||
|
||||
# --- Groups (Ko-Residenz) ----------------------------------------------------
|
||||
def set_group(group: str, members: list[str], swap: bool = False, persist: bool = False) -> None:
|
||||
"""llama-swap-`groups`-Eintrag setzen. swap=False → alle Mitglieder dürfen
|
||||
GLEICHZEITIG laufen (Ko-Residenz, keine Nachlade-Latenz). persist=True →
|
||||
Mitglieder werden nie automatisch entladen."""
|
||||
cfg = read_config()
|
||||
groups = cfg.setdefault("groups", {})
|
||||
groups[group] = {"swap": swap, "persist": persist, "members": list(members)}
|
||||
write_config(cfg)
|
||||
|
||||
|
||||
def list_groups() -> dict:
|
||||
return read_config().get("groups") or {}
|
||||
|
||||
|
||||
def set_role(model_id: str, role: str | None) -> bool:
|
||||
"""Rolle (llama-swap-Alias) eines bestehenden Modells setzen/ändern. So tauscht man
|
||||
z.B. das `fast`-Hirn: Rolle `fast` auf ein anderes Modell legen (Alias wandert)."""
|
||||
cfg = read_config()
|
||||
if model_id not in (cfg.get("models") or {}):
|
||||
return False
|
||||
set_role_alias(cfg, model_id, role)
|
||||
write_config(cfg)
|
||||
return True
|
||||
|
||||
|
||||
def set_ctx(model_id: str, ctx: int) -> bool:
|
||||
"""Kontextlänge (-c) eines bestehenden Modells ändern."""
|
||||
cfg = read_config()
|
||||
spec = (cfg.get("models") or {}).get(model_id)
|
||||
if not spec:
|
||||
return False
|
||||
cmd = str(spec.get("cmd", ""))
|
||||
if _CTX_RE.search(cmd):
|
||||
cmd = re.sub(r"-(?:c|-ctx-size)\s+\d+", f"-c {ctx}", cmd)
|
||||
else:
|
||||
cmd = cmd.rstrip() + f" -c {ctx}"
|
||||
spec["cmd"] = LiteralScalarString(cmd if cmd.endswith("\n") else cmd + "\n")
|
||||
write_config(cfg)
|
||||
return True
|
||||
|
||||
|
||||
def set_ttl(model_id: str, ttl: int) -> bool:
|
||||
"""Idle-TTL (Sekunden) eines bestehenden Modells setzen. ttl=0 → nie automatisch
|
||||
entladen (für das Agent-Hirn, das dauerhaft warm bleiben muss)."""
|
||||
cfg = read_config()
|
||||
spec = (cfg.get("models") or {}).get(model_id)
|
||||
if not isinstance(spec, dict):
|
||||
return False
|
||||
spec["ttl"] = int(ttl)
|
||||
write_config(cfg)
|
||||
return True
|
||||
|
||||
|
||||
def delete_model(model_id: str) -> bool:
|
||||
"""Entfernt einen Modell-Eintrag aus der config.yaml, löscht die zugehörigen
|
||||
GGUF-Dateien (auch Splits) vom Datenträger und bereinigt leere Ordner.
|
||||
"""
|
||||
cfg = read_config()
|
||||
models = cfg.get("models") or {}
|
||||
if model_id not in models:
|
||||
return False
|
||||
|
||||
model_spec = models[model_id] or {}
|
||||
cmd = str(model_spec.get("cmd", "")).strip()
|
||||
if (m := _PATH_RE.search(cmd)):
|
||||
path = m.group(1).replace("'", "").replace('"', "")
|
||||
if path:
|
||||
# 1. Haupt-GGUF-Datei löschen
|
||||
if os.path.exists(path):
|
||||
try:
|
||||
os.remove(path)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 2. Split-GGUF-Teile löschen (z.B. dateiname-00001-of-00005.gguf etc.)
|
||||
dirname = os.path.dirname(path)
|
||||
basename = os.path.basename(path)
|
||||
if os.path.isdir(dirname):
|
||||
split_idx = basename.find("-00001-of-")
|
||||
if split_idx != -1:
|
||||
prefix = basename[:split_idx]
|
||||
for f in os.listdir(dirname):
|
||||
if f.startswith(prefix) and f.endswith(".gguf"):
|
||||
try:
|
||||
os.remove(os.path.join(dirname, f))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# mmproj-Datei (Vision adapter) aus dem Befehl parsen & löschen
|
||||
if "mmproj" in cmd:
|
||||
mmproj_match = re.search(r'--mmproj\s+[\'"]?([^\s\'"]+)[\'"]?', cmd)
|
||||
if mmproj_match:
|
||||
m_path = mmproj_match.group(1)
|
||||
if os.path.exists(m_path):
|
||||
try:
|
||||
os.remove(m_path)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# 3. Eltern-Ordner löschen, falls er leer ist und nicht der Modelle-Wurzelordner selbst ist
|
||||
try:
|
||||
if not os.listdir(dirname) and os.path.basename(dirname) != "models":
|
||||
os.rmdir(dirname)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
del models[model_id]
|
||||
for g in (cfg.get("groups") or {}).values():
|
||||
if isinstance(g, dict) and model_id in (g.get("members") or []):
|
||||
g["members"] = [m for m in g["members"] if m != model_id]
|
||||
write_config(cfg)
|
||||
return True
|
||||
|
||||
|
||||
def brain_model_name() -> str | None:
|
||||
"""Modellname von Lucys Agent-Hirn. Bevorzugt das Modell mit dem 'hermes'-Alias/-Rolle;
|
||||
fällt auf Hermes' aktives `model.default` zurück (deckt den Fall ab, dass die Config direkt
|
||||
auf einen Modellnamen statt den Alias zeigt)."""
|
||||
models = list_models()
|
||||
for m in models:
|
||||
names = {str(a).lower() for a in (m.get("aliases") or [])}
|
||||
if m.get("role"):
|
||||
names.add(str(m["role"]).lower())
|
||||
if "hermes" in names:
|
||||
return m["name"]
|
||||
# Fallback: das real von Hermes genutzte Hirn (model.default), per Alias/Name auflösen.
|
||||
try:
|
||||
from services.agent import _active_brain_name
|
||||
brain = (_active_brain_name() or "").lower()
|
||||
if brain and brain != "auto":
|
||||
cur = next((m for m in models if (m.get("role") or "").lower() == brain), None) \
|
||||
or next((m for m in models if brain in (m["name"] or "").lower()), None)
|
||||
if cur:
|
||||
return cur["name"]
|
||||
except Exception:
|
||||
log.debug("brain_model_name: Hermes-Fallback fehlgeschlagen", exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def brain_status() -> dict:
|
||||
"""Ist Lucys Agent-Hirn (Rolle 'hermes') WIRKLICH geladen & bereit? Prüft /running — ein
|
||||
abgestürztes Modell (z.B. OOM/Crash nach Engine-Update) erscheint dort NICHT als running.
|
||||
Fängt damit den Fall 'Engine erreichbar, aber Hirn tot', den engine_reachable() nicht sieht."""
|
||||
name = brain_model_name()
|
||||
running = get_running_models()
|
||||
return {"role": "hermes", "model": name, "ready": bool(name and name in running)}
|
||||
|
||||
|
||||
def get_running_models() -> list[str]:
|
||||
"""Fragt den /running Endpunkt von llama-swap ab. Gibt die Namen der geladenen
|
||||
Modelle zurück. Neuere llama-swap-Versionen liefern Objekte ({model, state, ...})
|
||||
statt Strings — beide Formen werden auf Namens-Strings normalisiert."""
|
||||
try:
|
||||
with httpx.Client(timeout=2.0) as c:
|
||||
r = c.get(f"{LLAMA_SWAP_URL}/running")
|
||||
if r.status_code == 200:
|
||||
data = r.json().get("running") or []
|
||||
return [x.get("model", "") if isinstance(x, dict) else x for x in data]
|
||||
except Exception:
|
||||
log.warning("get_running_models fehlgeschlagen", exc_info=True)
|
||||
return []
|
||||
@@ -0,0 +1,568 @@
|
||||
"""
|
||||
Wartung: Updates (OS/Engine/Modelle), Dienst-Neustart (system- vs user-aware),
|
||||
Reboot, Logs. Portiert/modernisiert aus Mission Control v1 (routers/maintenance.py).
|
||||
|
||||
Passwortfrei über NOPASSWD-Whitelist (sudo -n). OS-Update/Reboot brauchen einmalig
|
||||
erweiterte sudoers (siehe docs/BEDIENUNG.md). Lange Ops laufen als jobengine-Job.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import time
|
||||
from datetime import datetime
|
||||
|
||||
import httpx
|
||||
import psutil
|
||||
|
||||
from services import catalog, discover, jobengine, llamaswap, system
|
||||
|
||||
# System-Dienste (root, via sudo -n NOPASSWD) vs. User-Dienste (systemctl --user).
|
||||
SYSTEM_SERVICES = {"llama-swap"}
|
||||
USER_SERVICES = {"mission-control-2", "hermes-gateway", "hermes-terminal", "mem0-service", "voice-service"}
|
||||
|
||||
# Engine-Update: lädt den neuesten Vulkan-Build (deploy/update-engine.sh, läuft als root).
|
||||
_REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", ".."))
|
||||
ENGINE_UPDATE_CMD = os.environ.get(
|
||||
"MC_ENGINE_UPDATE_CMD", f"sudo bash {_REPO_ROOT}/deploy/update-engine.sh")
|
||||
# Engine = offizieller Vulkan-Build von ggml-org/llama.cpp (RADV auf Strix Halo).
|
||||
ENGINE_PATH = os.environ.get("MC_ENGINE_PATH", "/opt/llamacpp-vulkan")
|
||||
ENGINE_REPO = os.environ.get("MC_ENGINE_REPO", "ggml-org/llama.cpp")
|
||||
_engine_cache = {"ts": 0.0, "avail": False}
|
||||
|
||||
# Router = llama-swap (mostlygeek): proxyt Anfragen und wechselt die Modelle heiß. Eigenes
|
||||
# Upstream-Projekt mit eigenem Release-Zyklus → getrennt von der Engine geführt.
|
||||
SWAP_UPDATE_CMD = os.environ.get(
|
||||
"MC_SWAP_UPDATE_CMD", f"sudo bash {_REPO_ROOT}/deploy/update-swap.sh")
|
||||
SWAP_BIN = os.environ.get("MC_SWAP_BIN", "/usr/local/bin/llama-swap")
|
||||
SWAP_REPO = os.environ.get("MC_SWAP_REPO", "mostlygeek/llama-swap")
|
||||
_swap_cache = {"ts": 0.0, "avail": False}
|
||||
|
||||
# Stack-Funktionsprüfung NACH jedem Update (OS/Engine/Router): verifiziert per echter
|
||||
# Inferenz, dass der Stack noch läuft → Job wird rot, wenn ein Update etwas zerschossen hat.
|
||||
STACK_POSTCHECK = os.path.join(_REPO_ROOT, "deploy", "stack-postcheck.sh")
|
||||
|
||||
|
||||
def _installed_engine_build() -> int | None:
|
||||
"""Build-Nummer der installierten llama-server-Binary (z.B. 9821), oder None.
|
||||
Vulkan-Build braucht LD_LIBRARY_PATH=ENGINE_PATH zum Start von --version."""
|
||||
bin_path = os.path.join(ENGINE_PATH, "llama-server")
|
||||
if not os.path.exists(bin_path):
|
||||
return None
|
||||
try:
|
||||
env = dict(os.environ, LD_LIBRARY_PATH=ENGINE_PATH)
|
||||
out = subprocess.run([bin_path, "--version"], capture_output=True, text=True,
|
||||
timeout=20, env=env)
|
||||
txt = (out.stderr or "") + (out.stdout or "")
|
||||
# Formate je nach Build: "version: 9821 (hash)" (aktuell), "build: <hash> (9821)", "b9821".
|
||||
if (m := re.search(r"version:\s*(\d{3,})", txt)) \
|
||||
or (m := re.search(r"build:\s*\S+\s*\((\d+)\)", txt)) \
|
||||
or (m := re.search(r"\bb(\d{3,})\b", txt)):
|
||||
return int(m.group(1))
|
||||
except Exception:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _ram_gb() -> float:
|
||||
return psutil.virtual_memory().total / (1024 ** 3)
|
||||
|
||||
|
||||
def _os_upgradable() -> int:
|
||||
try:
|
||||
# LC_ALL=C erzwingt englische apt-Ausgabe ("[upgradable from: ...]") — sonst zählt
|
||||
# grep auf einer deutschen Box ("[aktualisierbar von:]") nichts und meldet faelschlich 0.
|
||||
out = subprocess.run(
|
||||
["bash", "-c", "LC_ALL=C apt list --upgradable 2>/dev/null | grep -c upgradable || true"],
|
||||
capture_output=True, text=True, timeout=10)
|
||||
return int((out.stdout or "0").strip() or 0)
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
|
||||
def _engine_update_available() -> bool:
|
||||
now = time.time()
|
||||
if now - _engine_cache["ts"] < 3600:
|
||||
return _engine_cache["avail"]
|
||||
avail = False
|
||||
try:
|
||||
rel = httpx.get(f"https://api.github.com/repos/{ENGINE_REPO}/releases/latest",
|
||||
timeout=6, headers={"User-Agent": "MissionControl2"}).json()
|
||||
tag = str(rel.get("tag_name", ""))
|
||||
latest = int(m.group(1)) if (m := re.search(r"(\d{3,})", tag)) else None
|
||||
installed = _installed_engine_build()
|
||||
if latest is not None and installed is not None:
|
||||
avail = latest > installed # präziser Build-Nummer-Vergleich
|
||||
else: # Fallback: Release-Datum vs. Engine-mtime
|
||||
pub = datetime.fromisoformat(rel["published_at"].replace("Z", "+00:00")).timestamp()
|
||||
avail = pub > os.path.getmtime(ENGINE_PATH) + 86400
|
||||
except Exception:
|
||||
avail = False
|
||||
_engine_cache.update(ts=now, avail=avail)
|
||||
return avail
|
||||
|
||||
|
||||
def _installed_swap_version() -> int | None:
|
||||
"""Versions-Nummer der installierten llama-swap-Binary (z.B. 228), oder None."""
|
||||
if not os.path.exists(SWAP_BIN):
|
||||
return None
|
||||
try:
|
||||
out = subprocess.run([SWAP_BIN, "--version"], capture_output=True, text=True, timeout=15)
|
||||
txt = (out.stdout or "") + (out.stderr or "")
|
||||
if (m := re.search(r"version:\s*(\d+)", txt)) or (m := re.search(r"\bv?(\d{2,})\b", txt)):
|
||||
return int(m.group(1))
|
||||
except Exception:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _swap_update_available() -> bool:
|
||||
now = time.time()
|
||||
if now - _swap_cache["ts"] < 3600:
|
||||
return _swap_cache["avail"]
|
||||
avail = False
|
||||
try:
|
||||
rel = httpx.get(f"https://api.github.com/repos/{SWAP_REPO}/releases/latest",
|
||||
timeout=6, headers={"User-Agent": "MissionControl2"}).json()
|
||||
tag = str(rel.get("tag_name", ""))
|
||||
latest = int(m.group(1)) if (m := re.search(r"(\d{2,})", tag)) else None
|
||||
installed = _installed_swap_version()
|
||||
if latest is not None and installed is not None:
|
||||
avail = latest > installed
|
||||
except Exception:
|
||||
avail = False
|
||||
_swap_cache.update(ts=now, avail=avail)
|
||||
return avail
|
||||
|
||||
|
||||
_comp_cache = {"ts": 0.0, "data": []}
|
||||
|
||||
|
||||
def _hermes_agent_update() -> dict:
|
||||
"""Hermes-Agent wird aus **git** aktualisiert (CLI `hermes update` = git pull origin <branch>).
|
||||
Darum HEAD vs. origin/<branch> prüfen (fetch + behind-count) — NICHT GitHub-Releases: die
|
||||
werden selten getaggt, main läuft ihnen voraus → sonst zeigt das UI nie ein Update an."""
|
||||
info = {"key": "hermes_agent", "name": "Hermes Agent", "current": None,
|
||||
"latest": None, "update": False, "reachable": None}
|
||||
git = system.find_hermes_agent_git()
|
||||
if not git or not git.get("path"):
|
||||
return info
|
||||
path = git["path"]
|
||||
info["current"] = git.get("hash")
|
||||
try:
|
||||
branch = (subprocess.run(["git", "-C", path, "rev-parse", "--abbrev-ref", "HEAD"],
|
||||
capture_output=True, text=True, timeout=8).stdout.strip() or "main")
|
||||
fetch = subprocess.run(["git", "-C", path, "fetch", "-q", "origin", branch],
|
||||
capture_output=True, text=True, timeout=25)
|
||||
info["reachable"] = (fetch.returncode == 0)
|
||||
if fetch.returncode == 0:
|
||||
cnt = subprocess.run(["git", "-C", path, "rev-list", "--count", f"HEAD..origin/{branch}"],
|
||||
capture_output=True, text=True, timeout=8)
|
||||
behind = int(cnt.stdout.strip() or "0") if cnt.returncode == 0 else 0
|
||||
info["behind"] = behind
|
||||
info["update"] = behind > 0
|
||||
oh = subprocess.run(["git", "-C", path, "rev-parse", "--short", f"origin/{branch}"],
|
||||
capture_output=True, text=True, timeout=8).stdout.strip()
|
||||
info["latest"] = (f"{oh} ({behind} neu)" if behind else (oh or info["current"]))
|
||||
except Exception:
|
||||
info["reachable"] = False
|
||||
return info
|
||||
|
||||
|
||||
def _components_cached() -> list[dict]:
|
||||
"""Update-Status von Hermes-Agent (1h-Cache → GitHub schonen)."""
|
||||
now = time.time()
|
||||
if now - _comp_cache["ts"] < 3600 and _comp_cache["data"]:
|
||||
return _comp_cache["data"]
|
||||
data = [_hermes_agent_update()]
|
||||
_comp_cache.update(ts=now, data=data)
|
||||
return data
|
||||
|
||||
|
||||
def _params_of(m: dict) -> float:
|
||||
"""Größen-bewusste Parameterzahl eines installierten Modells: max aus Namens-Schätzung
|
||||
und Dateigröße (fängt namenlose wie 'Qwen3-Coder-Next' UND Split-GGUFs ab)."""
|
||||
from services.fit import QUANT_BYTES_PER_PARAM
|
||||
caps = m.get("capabilities") or {}
|
||||
bpp = QUANT_BYTES_PER_PARAM.get((m.get("quant") or "Q4_K_M").upper(), 0.55)
|
||||
size_gb = (m.get("size_bytes") or 0) / (1024 ** 3)
|
||||
pb_size = (size_gb / bpp) if size_gb > 1.0 else 0.0
|
||||
return max(float(caps.get("params_b") or 0), pb_size, 0.0)
|
||||
|
||||
|
||||
# Familien-Subtyp + Generations-Version aus dem Modellnamen (für „echtes Upgrade?").
|
||||
_FAM_PATS = (("qwen", r"qwen(\d+(?:\.\d+)?)"), ("gemma", r"gemma[-_ ]?(\d+(?:\.\d+)?)"),
|
||||
("llama", r"llama[-_ ]?(\d+(?:\.\d+)?)"), ("phi", r"phi[-_ ]?(\d+(?:\.\d+)?)"),
|
||||
("mistral", r"mistral"), ("hermes", r"hermes[-_ ]?(\d+(?:\.\d+)?)"))
|
||||
|
||||
|
||||
def _gen_key(name: str):
|
||||
"""(Familie+Subtyp, Generations-Version) oder None. Z.B. 'Qwen3-VL-2B' → ('qwen-vl', 3.0),
|
||||
'Qwen2.5-VL-7B' → ('qwen-vl', 2.5). Nur gleiche Familie ist sinnvoll vergleichbar."""
|
||||
low = (name or "").lower()
|
||||
sub = "-vl" if any(k in low for k in ("-vl", "vl-", "vision", "llava", "pixtral")) else \
|
||||
"-coder" if ("coder" in low or "-code" in low) else ""
|
||||
for fam, pat in _FAM_PATS:
|
||||
m = re.search(pat, low)
|
||||
if m:
|
||||
ver = float(m.group(1)) if (m.groups() and m.group(1)) else 0.0
|
||||
return (fam + sub, ver)
|
||||
return None
|
||||
|
||||
|
||||
def _meta(name: str, model_dict: dict | None = None, im: dict | None = None) -> dict:
|
||||
"""Metadaten (family, gen, total, active, moe) — bevorzugt den kuratierten Katalog,
|
||||
sonst die Felder eines Discover-/Modell-Dicts, sonst Namens-/Größen-Heuristik."""
|
||||
cm = catalog.meta_for_name(name)
|
||||
if cm:
|
||||
return {"family": cm.get("family"), "gen": cm.get("generation"),
|
||||
"total": float(cm.get("total_params_b") or 0),
|
||||
"active": cm.get("active_params_b"), "moe": bool(cm.get("moe"))}
|
||||
d = model_dict or {}
|
||||
g = _gen_key(name)
|
||||
total = float(d.get("params_b") or 0) or (_params_of(im) if im else 0.0)
|
||||
return {"family": (d.get("family") or (g[0] if g else None)),
|
||||
"gen": (d.get("generation") if d.get("generation") is not None else (g[1] if g else None)),
|
||||
"total": total, "active": d.get("active_b"), "moe": bool(d.get("moe"))}
|
||||
|
||||
|
||||
def model_upgrades() -> list[dict]:
|
||||
"""Je Rolle ein ECHTES Upgrade — nur wenn die Empfehlung wirklich besser ist:
|
||||
gleiche Familie UND (neuere Generation ODER deutlich größer) UND kein Tempo-Downgrade
|
||||
(MoE-first für die bandbreiten-limitierte Box: dense ersetzt MoE nur bei großem Wissens-
|
||||
Sprung). Metadaten kommen aus dem kuratierten Katalog → keine Namens-Raterei."""
|
||||
disc = discover.safe_discover(_ram_gb())
|
||||
if not disc:
|
||||
return []
|
||||
|
||||
installed = llamaswap.list_models()
|
||||
inst_by_role = {m["role"]: m for m in installed if m.get("role")}
|
||||
cmds = " ".join(str(s.get("cmd", "")).lower()
|
||||
for s in (llamaswap.read_config().get("models") or {}).values())
|
||||
out = []
|
||||
|
||||
for c in disc.get("categories", []):
|
||||
role = c["role"]
|
||||
im = inst_by_role.get(role)
|
||||
if im is None:
|
||||
continue
|
||||
rec = c.get("recommended")
|
||||
if not rec:
|
||||
continue
|
||||
rec_model = next((x for x in c.get("models", []) if x.get("repo") == rec), None)
|
||||
|
||||
i = _meta(im["name"], im=im)
|
||||
r = _meta(rec, model_dict=rec_model)
|
||||
|
||||
if not i["family"] or not r["family"] or i["family"] != r["family"]:
|
||||
continue # andere/unbekannte Familie → kein Upgrade
|
||||
if r["gen"] is not None and i["gen"] is not None and r["gen"] < i["gen"] - 1e-6:
|
||||
continue # ältere Generation → niemals
|
||||
same_gen = (r["gen"] is None or i["gen"] is None or abs(r["gen"] - i["gen"]) < 1e-6)
|
||||
if same_gen:
|
||||
if r["total"] and i["total"] and r["total"] < i["total"] * 1.05:
|
||||
continue # gleiche Gen, nicht größer → kein Upgrade
|
||||
# MoE-first: ein MoE durch dense ersetzen nur bei deutlichem Wissens-Sprung
|
||||
if i["moe"] and not r["moe"] and r["total"] < i["total"] * 1.5:
|
||||
continue
|
||||
# Tempo nicht verschlechtern (aktive Params), außer großer Wissens-Gewinn
|
||||
ia, ra = (i["active"] or i["total"]), (r["active"] or r["total"])
|
||||
if ia and ra > ia * 1.3 and r["total"] < i["total"] * 1.3:
|
||||
continue
|
||||
|
||||
base = rec.split("/")[-1].lower()
|
||||
stem = base[:-5] if base.endswith("-gguf") else base
|
||||
if base in cmds or (stem and stem in cmds):
|
||||
continue # schon installiert
|
||||
out.append({"role": role, "title": c["title"], "repo": rec})
|
||||
return out
|
||||
|
||||
|
||||
def _last_apt_update() -> float | None:
|
||||
for path in ["/var/lib/apt/periodic/update-success-stamp", "/var/cache/apt/pkgcache.bin"]:
|
||||
if os.path.exists(path):
|
||||
try:
|
||||
return os.path.getmtime(path)
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def updates() -> dict:
|
||||
ups = model_upgrades()
|
||||
return {"os": _os_upgradable(), "engine": 1 if _engine_update_available() else 0,
|
||||
"swap": 1 if _swap_update_available() else 0,
|
||||
"models": len(ups), "model_list": ups, "last_check": _last_apt_update(),
|
||||
"components": _components_cached()}
|
||||
|
||||
|
||||
# ── Update-Details (was genau wird aktualisiert) — on-demand beim Öffnen des Fensters ──
|
||||
|
||||
def os_update_details() -> dict:
|
||||
"""Liste der aktualisierbaren apt-Pakete (Name, installiert → Kandidat)."""
|
||||
out_pkgs: list[dict] = []
|
||||
try:
|
||||
# LC_ALL=C → englische Ausgabe, damit der Regex "[upgradable from: ...]" greift
|
||||
# (deutsche Box meldet sonst "[aktualisierbar von:]" und die Liste bliebe leer).
|
||||
out = subprocess.run(["bash", "-c", "LC_ALL=C apt list --upgradable 2>/dev/null"],
|
||||
capture_output=True, text=True, timeout=20)
|
||||
for line in (out.stdout or "").splitlines():
|
||||
# Format: name/repo neue_version arch [upgradable from: alte_version]
|
||||
m = re.match(r"^([^/\s]+)/\S+\s+(\S+)\s+\S+\s+\[upgradable from:\s*([^\]]+)\]",
|
||||
line.strip())
|
||||
if m:
|
||||
out_pkgs.append({"name": m.group(1), "candidate": m.group(2),
|
||||
"current": m.group(3).strip()})
|
||||
out_pkgs.sort(key=lambda p: p["name"])
|
||||
return {"kind": "os", "count": len(out_pkgs), "packages": out_pkgs}
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {"kind": "os", "count": len(out_pkgs), "packages": out_pkgs, "error": str(exc)}
|
||||
|
||||
|
||||
def engine_update_details() -> dict:
|
||||
"""Installierte vs. neueste Engine-Build-Nummer + Release-Name/-Notizen/-Link."""
|
||||
info: dict = {"kind": "engine", "installed_build": _installed_engine_build(),
|
||||
"latest_build": None, "latest_tag": None, "name": None,
|
||||
"url": None, "body": None}
|
||||
try:
|
||||
rel = httpx.get(f"https://api.github.com/repos/{ENGINE_REPO}/releases/latest",
|
||||
timeout=8, headers={"User-Agent": "MissionControl2"}).json()
|
||||
tag = str(rel.get("tag_name", ""))
|
||||
info["latest_tag"] = tag
|
||||
info["latest_build"] = int(m.group(1)) if (m := re.search(r"(\d{3,})", tag)) else None
|
||||
info["name"] = rel.get("name") or tag
|
||||
info["url"] = rel.get("html_url")
|
||||
body = (rel.get("body") or "").strip()
|
||||
info["body"] = body[:2000] if body else None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
info["error"] = str(exc)
|
||||
return info
|
||||
|
||||
|
||||
def swap_update_details() -> dict:
|
||||
"""Installierte vs. neueste llama-swap-Version + Release-Name/-Notizen/-Link."""
|
||||
info: dict = {"kind": "swap", "installed_build": _installed_swap_version(),
|
||||
"latest_build": None, "latest_tag": None, "name": None,
|
||||
"url": None, "body": None}
|
||||
try:
|
||||
rel = httpx.get(f"https://api.github.com/repos/{SWAP_REPO}/releases/latest",
|
||||
timeout=8, headers={"User-Agent": "MissionControl2"}).json()
|
||||
tag = str(rel.get("tag_name", ""))
|
||||
info["latest_tag"] = tag
|
||||
info["latest_build"] = int(m.group(1)) if (m := re.search(r"(\d{2,})", tag)) else None
|
||||
info["name"] = rel.get("name") or tag
|
||||
info["url"] = rel.get("html_url")
|
||||
body = (rel.get("body") or "").strip()
|
||||
info["body"] = body[:2000] if body else None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
info["error"] = str(exc)
|
||||
return info
|
||||
|
||||
|
||||
def hermes_update_details() -> dict:
|
||||
"""Commits, die ein Hermes-Update einspielen würde (HEAD..origin/<branch>)."""
|
||||
info: dict = {"kind": "hermes", "branch": None, "behind": 0, "commits": []}
|
||||
git = system.find_hermes_agent_git()
|
||||
if not git or not git.get("path"):
|
||||
info["error"] = "Hermes-Agent-Repo nicht gefunden."
|
||||
return info
|
||||
path = git["path"]
|
||||
try:
|
||||
branch = (subprocess.run(["git", "-C", path, "rev-parse", "--abbrev-ref", "HEAD"],
|
||||
capture_output=True, text=True, timeout=8).stdout.strip() or "main")
|
||||
info["branch"] = branch
|
||||
subprocess.run(["git", "-C", path, "fetch", "-q", "origin", branch],
|
||||
capture_output=True, text=True, timeout=25)
|
||||
log = subprocess.run(["git", "-C", path, "log", "--pretty=format:%h\x1f%s\x1f%cr",
|
||||
f"HEAD..origin/{branch}"], capture_output=True, text=True, timeout=10)
|
||||
commits = []
|
||||
for line in (log.stdout or "").splitlines():
|
||||
parts = line.split("\x1f")
|
||||
if len(parts) == 3:
|
||||
commits.append({"hash": parts[0], "subject": parts[1], "when": parts[2]})
|
||||
info["commits"] = commits
|
||||
info["behind"] = len(commits)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
info["error"] = str(exc)
|
||||
return info
|
||||
|
||||
|
||||
def update_details(kind: str) -> dict:
|
||||
return {"os": os_update_details, "engine": engine_update_details,
|
||||
"swap": swap_update_details,
|
||||
"hermes": hermes_update_details}.get(kind, lambda: {"error": "unbekannt"})()
|
||||
|
||||
|
||||
def _run(cmd: list[str], sudo_password: str | None = None) -> dict:
|
||||
actual_cmd = list(cmd)
|
||||
has_sudo = False
|
||||
|
||||
if cmd and cmd[0] == "sudo":
|
||||
has_sudo = True
|
||||
# If we have a password, use -S instead of -n
|
||||
if sudo_password is not None:
|
||||
if "-n" in actual_cmd:
|
||||
actual_cmd = [x for x in actual_cmd if x != "-n"]
|
||||
if "-S" not in actual_cmd:
|
||||
actual_cmd.insert(1, "-S")
|
||||
else:
|
||||
# Force -n to fail cleanly if password is required
|
||||
if "-S" in actual_cmd:
|
||||
actual_cmd = [x for x in actual_cmd if x != "-S"]
|
||||
if "-n" not in actual_cmd:
|
||||
actual_cmd.insert(1, "-n")
|
||||
|
||||
try:
|
||||
input_data = (sudo_password + "\n") if (has_sudo and sudo_password is not None) else None
|
||||
p = subprocess.run(actual_cmd, input=input_data, capture_output=True, text=True, timeout=120)
|
||||
|
||||
err_msg = p.stderr or ""
|
||||
if p.returncode != 0 and ("a password is required" in err_msg or "password" in err_msg.lower() or "sudo:" in err_msg):
|
||||
if sudo_password is not None:
|
||||
return {"ok": False, "status": "incorrect_password", "out": p.stdout or "", "err": "Falsches Sudo-Passwort."}
|
||||
return {"ok": False, "status": "password_required", "out": p.stdout or "", "err": "Sudo-Passwort erforderlich."}
|
||||
|
||||
return {"ok": p.returncode == 0, "out": (p.stdout or "")[-4000:], "err": (p.stderr or "")[-2000:]}
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {"ok": False, "out": "", "err": str(exc)}
|
||||
|
||||
|
||||
def check_sudo_needs_password(sudo_password: str | None = None) -> dict | None:
|
||||
"""Checks if sudo needs a password. Returns error dict if password required/incorrect, else None."""
|
||||
res = _run(["sudo", "true"], sudo_password=sudo_password)
|
||||
if not res["ok"]:
|
||||
return res
|
||||
return None
|
||||
|
||||
|
||||
def restart_service(name: str, sudo_password: str | None = None) -> dict:
|
||||
if name in SYSTEM_SERVICES:
|
||||
if err := check_sudo_needs_password(sudo_password):
|
||||
return err
|
||||
return _run(["sudo", "systemctl", "restart", name], sudo_password=sudo_password)
|
||||
if name in USER_SERVICES:
|
||||
return _run(["systemctl", "--user", "restart", name])
|
||||
return {"ok": False, "err": f"Dienst '{name}' nicht erlaubt."}
|
||||
|
||||
|
||||
def logs(service: str, lines: int = 200, sudo_password: str | None = None) -> dict:
|
||||
lines = max(1, min(lines, 1000))
|
||||
if service in USER_SERVICES:
|
||||
r = _run(["journalctl", "--user", "-u", service, "-n", str(lines), "--no-pager"])
|
||||
return {"ok": r["ok"], "text": r["out"] or r["err"]}
|
||||
if service in SYSTEM_SERVICES:
|
||||
# Journal-Lesen braucht i.d.R. KEIN sudo (User ist in Gruppe adm/systemd-journal).
|
||||
# Erst ohne sudo versuchen; nur bei fehlenden Rechten auf sudo zurückfallen.
|
||||
r = _run(["journalctl", "-u", service, "-n", str(lines), "--no-pager"])
|
||||
if r["ok"]:
|
||||
return {"ok": True, "text": r["out"] or "(keine Log-Einträge)"}
|
||||
if err := check_sudo_needs_password(sudo_password):
|
||||
return err
|
||||
r = _run(["sudo", "journalctl", "-u", service, "-n", str(lines), "--no-pager"],
|
||||
sudo_password=sudo_password)
|
||||
return {"ok": r["ok"], "text": r["out"] or r["err"]}
|
||||
return {"ok": False, "text": "", "err": "Dienst nicht erlaubt."}
|
||||
|
||||
def check_updates_job(sudo_password: str | None = None) -> dict:
|
||||
if err := check_sudo_needs_password(sudo_password):
|
||||
return err
|
||||
|
||||
def on_done():
|
||||
_engine_cache.update(ts=0.0, avail=False)
|
||||
_comp_cache.update(ts=0.0, data=[]) # Hermes-Status ebenfalls neu berechnen lassen
|
||||
|
||||
cmd = "sudo apt-get update"
|
||||
job_id = jobengine.start_job(["bash", "-c", cmd], "Nach Updates suchen", on_done=on_done, sudo_password=sudo_password)
|
||||
return {"ok": True, "job_id": job_id}
|
||||
|
||||
|
||||
def _maintenance_busy() -> dict | None:
|
||||
"""Wartungs-Riegel: nur EIN binär-/dienst-veränderndes Update gleichzeitig. Verhindert
|
||||
Doppelklick UND parallele Updates aus zwei Tabs/Sessions (racende .bak-Sicherung/Restarts)."""
|
||||
if j := jobengine.active_in_group("maintenance"):
|
||||
return {"ok": False, "status": "busy", "running": j.get("label")}
|
||||
return None
|
||||
|
||||
|
||||
def os_update_job(sudo_password: str | None = None) -> dict:
|
||||
if busy := _maintenance_busy():
|
||||
return busy
|
||||
if err := check_sudo_needs_password(sudo_password):
|
||||
return err
|
||||
# Nach dem apt-Upgrade den Stack funktional prüfen (Job wird rot, wenn etwas kaputt ging).
|
||||
cmd = ("sudo apt-get update && sudo DEBIAN_FRONTEND=noninteractive apt-get upgrade -y "
|
||||
f"&& bash {STACK_POSTCHECK}")
|
||||
job_id = jobengine.start_job(["bash", "-c", cmd], "OS-Update (apt)",
|
||||
group="maintenance", sudo_password=sudo_password)
|
||||
return {"ok": True, "job_id": job_id}
|
||||
|
||||
|
||||
def engine_update_job(sudo_password: str | None = None) -> dict | None:
|
||||
if not ENGINE_UPDATE_CMD:
|
||||
return None
|
||||
if busy := _maintenance_busy():
|
||||
return busy
|
||||
if err := check_sudo_needs_password(sudo_password):
|
||||
return err
|
||||
|
||||
def on_done():
|
||||
_engine_cache.update(ts=0.0, avail=False) # Cache leeren → frischer Build-Vergleich
|
||||
|
||||
# update-engine.sh sichert den alten Build, aktualisiert, startet llama-swap neu, prüft den
|
||||
# Stack (stack-postcheck.sh) und rollt bei Fehler selbst zurück. Exit 0 nur bei verifiziertem
|
||||
# neuen Build → on_done (Cache leeren) läuft nur dann; bei Rollback (Exit 1/2) bleibt das Badge.
|
||||
job_id = jobengine.start_job(["bash", "-c", ENGINE_UPDATE_CMD],
|
||||
"Engine-Update (llama.cpp Vulkan)",
|
||||
group="maintenance", on_done=on_done, sudo_password=sudo_password)
|
||||
return {"ok": True, "job_id": job_id}
|
||||
|
||||
|
||||
def swap_update_job(sudo_password: str | None = None) -> dict | None:
|
||||
if not SWAP_UPDATE_CMD:
|
||||
return None
|
||||
if busy := _maintenance_busy():
|
||||
return busy
|
||||
if err := check_sudo_needs_password(sudo_password):
|
||||
return err
|
||||
|
||||
def on_done():
|
||||
_swap_cache.update(ts=0.0, avail=False) # Cache leeren → frischer Versions-Vergleich
|
||||
|
||||
# update-swap.sh sichert die alte Binary, aktualisiert, startet llama-swap neu, prüft den Stack
|
||||
# (stack-postcheck.sh) und rollt bei Fehler selbst zurück. Exit 0 nur bei verifizierter neuer
|
||||
# Version → on_done (Cache leeren) läuft nur dann; bei Rollback (Exit 1/2) bleibt das Badge.
|
||||
job_id = jobengine.start_job(["bash", "-c", SWAP_UPDATE_CMD],
|
||||
"Router-Update (llama-swap)",
|
||||
group="maintenance", on_done=on_done, sudo_password=sudo_password)
|
||||
return {"ok": True, "job_id": job_id}
|
||||
|
||||
|
||||
def hermes_update_job() -> dict:
|
||||
"""Hermes-Agent aktualisieren wie die CLI (`hermes update` = git pull + Deps), danach
|
||||
den Gateway neu starten. Davor ein Sicherheits-Backup (unser deploy/backup.sh). Kein sudo
|
||||
(alles im User-Space). Läuft als Hintergrund-Job (kann ~1 Min dauern)."""
|
||||
if busy := _maintenance_busy():
|
||||
return busy
|
||||
git = system.find_hermes_agent_git()
|
||||
path = (git or {}).get("path") or os.path.expanduser("~/.hermes/hermes-agent")
|
||||
py = os.path.join(path, "venv", "bin", "python")
|
||||
backup = os.path.join(_REPO_ROOT, "deploy", "backup.sh")
|
||||
postcheck = os.path.join(_REPO_ROOT, "deploy", "hermes-postcheck.sh")
|
||||
# Backup → update → Gateway-Neustart → Gehirn-Check (Job wird rot, wenn Mem0/Plugin kaputt).
|
||||
cmd = (f"bash {backup} || true; "
|
||||
f"cd {path} && {py} -m hermes_cli.main update --yes "
|
||||
f"&& systemctl --user restart hermes-gateway "
|
||||
f"&& sleep 4 && bash {postcheck}")
|
||||
|
||||
def on_done():
|
||||
_comp_cache.update(ts=0.0, data=[]) # Update-Status neu berechnen lassen
|
||||
|
||||
job_id = jobengine.start_job(["bash", "-c", cmd], "Hermes-Agent-Update",
|
||||
group="maintenance", on_done=on_done)
|
||||
return {"ok": True, "job_id": job_id}
|
||||
|
||||
|
||||
def reboot(sudo_password: str | None = None) -> dict:
|
||||
if err := check_sudo_needs_password(sudo_password):
|
||||
return err
|
||||
return _run(["sudo", "reboot"], sudo_password=sudo_password)
|
||||
@@ -0,0 +1,147 @@
|
||||
"""
|
||||
Geteiltes Gedächtnis (die „Verfassung") — jetzt auto-lernend & semantisch über Mem0.
|
||||
|
||||
Dieser Service ist nur noch ein dünner HTTP-Client auf den Mem0-Sidecar (mem0_service/app.py,
|
||||
läuft im ~/.mem0/venv unter Python 3.12). Die `/api/memory`-API-Form bleibt unverändert, damit
|
||||
UI und MCP-Server kompatibel bleiben. Neu gegenüber der alten flachen SQLite:
|
||||
|
||||
- search (q gesetzt) ist SEMANTISCH (Vektor/Embeddings) statt LIKE-Textsuche, mit Relevanz-Score.
|
||||
- learn() reicht Gesprächs-Turns durch → Mem0 EXTRAHIERT Fakten selbst (Auto-Lernen).
|
||||
- Dedup macht Mem0 beim Auto-Lernen selbst; der manuelle Kurator unten bleibt als Komfort.
|
||||
|
||||
5 Kategorien (user · instruction · stable · versioned · ephemeral) bleiben als Metadaten erhalten.
|
||||
"""
|
||||
|
||||
import re
|
||||
from difflib import SequenceMatcher
|
||||
|
||||
import httpx
|
||||
|
||||
from config import MEM0_SERVICE_URL
|
||||
|
||||
CATEGORIES = ("identity", "knowledge", "rules", "events")
|
||||
|
||||
_TIMEOUT = httpx.Timeout(60.0, connect=5.0) # LLM-Extraktion kann ein paar Sekunden dauern
|
||||
|
||||
|
||||
def _get(path: str, **params) -> list | dict:
|
||||
r = httpx.get(f"{MEM0_SERVICE_URL}{path}", params=params, timeout=_TIMEOUT)
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
|
||||
|
||||
def _post(path: str, data: dict) -> dict:
|
||||
r = httpx.post(f"{MEM0_SERVICE_URL}{path}", json=data, timeout=_TIMEOUT)
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
|
||||
|
||||
def _put(path: str, data: dict) -> dict:
|
||||
r = httpx.put(f"{MEM0_SERVICE_URL}{path}", json=data, timeout=_TIMEOUT)
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
|
||||
|
||||
def _delete(path: str) -> dict:
|
||||
r = httpx.delete(f"{MEM0_SERVICE_URL}{path}", timeout=_TIMEOUT)
|
||||
r.raise_for_status()
|
||||
return r.json()
|
||||
|
||||
|
||||
def list_memories(q: str = "", category: str = "") -> list[dict]:
|
||||
"""Alle Fakten oder — wenn q gesetzt — die semantisch ähnlichsten (mit `score`)."""
|
||||
return _get("/memory", **{k: v for k, v in (("q", q), ("category", category)) if v})
|
||||
|
||||
|
||||
def add_memory(content: str, category: str = "stable", source: str = "manual") -> dict:
|
||||
"""Einen Fakt VERBATIM speichern (keine LLM-Umformung). Auto-Lernen → learn()."""
|
||||
return _post("/memory", {"content": content.strip(), "category": category, "source": source})
|
||||
|
||||
|
||||
def update_memory(mid: str, content: str | None = None, category: str | None = None) -> dict | None:
|
||||
try:
|
||||
return _put(f"/memory/{mid}", {"content": content, "category": category})
|
||||
except httpx.HTTPStatusError as exc:
|
||||
if exc.response.status_code == 404:
|
||||
return None
|
||||
raise
|
||||
|
||||
|
||||
def delete_memory(mid: str) -> bool:
|
||||
try:
|
||||
_delete(f"/memory/{mid}")
|
||||
return True
|
||||
except httpx.HTTPStatusError as exc:
|
||||
if exc.response.status_code == 404:
|
||||
return False
|
||||
raise
|
||||
|
||||
|
||||
def learn(text: str | None = None, messages: list[dict] | None = None,
|
||||
source: str = "auto", category: str = "stable") -> dict:
|
||||
"""Auto-Lernen: Text/Gesprächs-Turns durchreichen → Mem0 extrahiert die Fakten selbst."""
|
||||
return _post("/learn", {"text": text, "messages": messages,
|
||||
"source": source, "category": category})
|
||||
|
||||
|
||||
def graph(min_score: float = 0.45, top_k: int = 3) -> dict:
|
||||
"""Fakten als Ähnlichkeits-Graph (Knoten + semantische Kanten) für die UI-Visualisierung."""
|
||||
return _get("/graph", min_score=min_score, top_k=top_k)
|
||||
|
||||
|
||||
def export_text() -> dict:
|
||||
rows = sorted(list_memories(), key=lambda r: (r.get("category", ""), r.get("updated_at", "")))
|
||||
lines = ["# Mission Control — Gedächtnis\n"]
|
||||
current = ""
|
||||
for r in rows:
|
||||
if r.get("category") != current:
|
||||
current = r.get("category", "")
|
||||
lines.append(f"\n## {current}\n")
|
||||
src = r.get("source", "")
|
||||
when = (r.get("updated_at") or "")[:10]
|
||||
lines.append(f"- {r.get('content', '')} _(Quelle: {src}, {when})_")
|
||||
return {"text": "\n".join(lines), "count": len(rows)}
|
||||
|
||||
|
||||
# --- Manueller Kurator (deterministisch, kein LLM) ---------------------------
|
||||
def _norm(s: str) -> str:
|
||||
s = re.sub(r"[^\w\s]", " ", s.lower(), flags=re.UNICODE)
|
||||
return re.sub(r"\s+", " ", s).strip()
|
||||
|
||||
|
||||
def dedupe(apply: bool = False, threshold: float = 0.85) -> dict:
|
||||
"""Findet Dubletten (exakt/enthalten/ähnlich) je Kategorie, behält den längsten
|
||||
Eintrag. Mem0 dedupliziert beim Auto-Lernen schon semantisch — das hier ist der
|
||||
manuelle Komfort-Knopf fürs UI (z.B. nach vielen Verbatim-Importen)."""
|
||||
rows = sorted(list_memories(), key=lambda r: (-len(r.get("content", "")), r.get("created_at", "")))
|
||||
used: set[str] = set()
|
||||
groups: list[dict] = []
|
||||
for i, a in enumerate(rows):
|
||||
if a["id"] in used:
|
||||
continue
|
||||
na = _norm(a.get("content", ""))
|
||||
if not na:
|
||||
continue
|
||||
dups = []
|
||||
for b in rows[i + 1:]:
|
||||
if b["id"] in used or b.get("category") != a.get("category"):
|
||||
continue
|
||||
nb = _norm(b.get("content", ""))
|
||||
if not nb:
|
||||
continue
|
||||
if nb in na or na in nb or SequenceMatcher(None, na, nb).ratio() >= threshold:
|
||||
dups.append(b); used.add(b["id"])
|
||||
if dups:
|
||||
used.add(a["id"])
|
||||
groups.append({
|
||||
"keep": {"id": a["id"], "content": a.get("content"), "category": a.get("category")},
|
||||
"remove": [{"id": d["id"], "content": d.get("content")} for d in dups],
|
||||
})
|
||||
dup_count = sum(len(g["remove"]) for g in groups)
|
||||
removed = 0
|
||||
if apply:
|
||||
for g in groups:
|
||||
for d in g["remove"]:
|
||||
if delete_memory(d["id"]):
|
||||
removed += 1
|
||||
return {"groups": groups, "duplicate_count": dup_count, "removed": removed, "applied": apply}
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Kosten-/Ersparnis-Berechnung für die Token-Statistik (eine Quelle der Wahrheit).
|
||||
|
||||
Vergleicht die lokal verbrauchten Tokens gegen die Cloud-Listenpreise vergleichbarer
|
||||
Modellklassen (Stand Juni 2026, USD pro 1M Tokens, in/out) und liefert die so
|
||||
eingesparte Summe. Wird vom System-Router dünn aufgerufen.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
# Cloud-Listenpreise je Rolle/Modellklasse: (input_usd_per_1M, output_usd_per_1M).
|
||||
PRICING: dict[str, tuple[float, float]] = {
|
||||
"heavy": (15.0, 75.0),
|
||||
"coder": (3.0, 15.0),
|
||||
"hermes": (1.0, 5.0),
|
||||
"fast": (0.15, 0.60),
|
||||
"scout": (0.15, 0.60),
|
||||
"vision": (0.15, 0.60),
|
||||
"reasoning": (0.15, 0.60),
|
||||
}
|
||||
# Tarif für nicht zuordenbare Tokens (Default-/Fallback-Klasse).
|
||||
DEFAULT_RATE: tuple[float, float] = (0.15, 0.60)
|
||||
# Baseline/Legacy-Tokens (vor modellspezifischem Logging) am Premium-Tarif bewerten,
|
||||
# damit historische Ersparnis erhalten bleibt.
|
||||
BASELINE_RATE: tuple[float, float] = PRICING["heavy"]
|
||||
USD_TO_EUR = float(os.environ.get("MC_USD_TO_EUR", "0.92"))
|
||||
|
||||
|
||||
def compute_savings(stats: dict, role_map: dict[str, str | None]) -> dict:
|
||||
"""Aggregiert Tokens und berechnet die Cloud-Ersparnis.
|
||||
|
||||
role_map: Modell-/Alias-Name (lowercase) -> Rolle, zur Tarif-Auflösung.
|
||||
"""
|
||||
prompt = stats.get("prompt_tokens", 0)
|
||||
completion = stats.get("completion_tokens", 0)
|
||||
|
||||
modeled_p = modeled_c = 0
|
||||
saved_usd = 0.0
|
||||
for m_name, m_tokens in (stats.get("models") or {}).items():
|
||||
mp = m_tokens.get("prompt", 0)
|
||||
mc = m_tokens.get("completion", 0)
|
||||
modeled_p += mp
|
||||
modeled_c += mc
|
||||
role = role_map.get(m_name, m_name)
|
||||
rate_in, rate_out = PRICING.get(role, DEFAULT_RATE)
|
||||
saved_usd += (mp * rate_in + mc * rate_out) / 1_000_000.0
|
||||
|
||||
baseline_p = max(0, prompt - modeled_p)
|
||||
baseline_c = max(0, completion - modeled_c)
|
||||
saved_usd += (baseline_p * BASELINE_RATE[0] + baseline_c * BASELINE_RATE[1]) / 1_000_000.0
|
||||
|
||||
return {
|
||||
"prompt_tokens": prompt,
|
||||
"completion_tokens": completion,
|
||||
"total_tokens": prompt + completion,
|
||||
"saved_usd": round(saved_usd, 2),
|
||||
"saved_eur": round(saved_usd * USD_TO_EUR, 2),
|
||||
"pricing": {role: {"in": r[0], "out": r[1]} for role, r in PRICING.items()},
|
||||
}
|
||||
@@ -0,0 +1,143 @@
|
||||
"""
|
||||
Rollen-Empfehlung: welches INSTALLIERTE Modell passt am besten auf eine Serving-Rolle?
|
||||
Capability-getrieben (Vision/Coder/Tools/MoE aus services.caps) + setup-bewusster Fit
|
||||
(services.budget). Speist den 'Empfohlen'-Hinweis + Auto-Pick im Rollen-Zuweisungs-Modal.
|
||||
|
||||
EINE Quelle der Wahrheit mit der ctx-/Fit-Logik: nutzt budget.setup_aware_ctx_for_model
|
||||
und fit.evaluate_fit — dieselbe Mathematik wie Install-Automatik und Auto-ctx-Button.
|
||||
"""
|
||||
|
||||
import psutil
|
||||
|
||||
from services import budget, catalog, llamaswap
|
||||
from services.fit import evaluate_fit
|
||||
|
||||
|
||||
def _ram_gb() -> float:
|
||||
return psutil.virtual_memory().total / (1024 ** 3)
|
||||
|
||||
|
||||
def _capability_suit(role: str, caps: dict, name: str) -> float:
|
||||
"""0..1 — Capability-Eignung (HARTE Gates). 0 = grundsätzlich falsch für die Rolle.
|
||||
Größe/Tempo bewertet getrennt _pref(), damit z.B. 'fast' nicht das größte Modell zieht."""
|
||||
role = (role or "").lower()
|
||||
low = (name or "").lower()
|
||||
vision = bool(caps.get("vision"))
|
||||
coder = bool(caps.get("coder"))
|
||||
tools = caps.get("tools") != "no"
|
||||
|
||||
if role == "vision":
|
||||
if not vision:
|
||||
return 0.0 # harte Anforderung
|
||||
return 1.0 if ("vl" in low or "llava" in low or "pixtral" in low) else 0.7 # dediziert > omni
|
||||
if role == "coder":
|
||||
return 1.0 if coder else 0.4 # Coder-Modell Pflicht für Empfehlung
|
||||
if role == "hermes":
|
||||
# Agent-Hirn: Hermes-Familie am robustesten; sonst natives Tool-Calling Pflicht.
|
||||
if "hermes" in low:
|
||||
return 1.0
|
||||
return 0.6 if tools else 0.1
|
||||
if role == "fast":
|
||||
# Alltags-Hirn braucht zuverlässige Tools; Größe/Tempo macht _pref.
|
||||
return 1.0 if tools else 0.6
|
||||
if role == "heavy":
|
||||
return 1.0
|
||||
if role == "scout":
|
||||
return 0.9 if vision else 0.7
|
||||
return 0.5
|
||||
|
||||
|
||||
def _pref(role: str, params: float, tps: float) -> float:
|
||||
"""0..1 — rollengerechte GRÖSSEN-/TEMPO-Präferenz. 'fast' belohnt Tempo & Kleinheit,
|
||||
'heavy' Größe (Wissen), 'hermes' moderate Größe (muss warm + ko-resident bleiben)."""
|
||||
role = (role or "").lower()
|
||||
if role == "fast":
|
||||
speed = min(tps / 25.0, 1.0)
|
||||
size_ok = 1.0 if params <= 50 else 50.0 / params
|
||||
return speed * size_ok
|
||||
if role == "heavy":
|
||||
return min(params / 120.0, 1.0)
|
||||
if role == "hermes":
|
||||
return 1.0 if params <= 24 else max(0.15, 24.0 / params) # 7–24B ideal als Hirn
|
||||
if role == "vision":
|
||||
return 1.0 if params <= 12 else 0.7 # klein/günstig bevorzugt
|
||||
if role == "coder":
|
||||
return 0.5 + 0.5 * min(params / 80.0, 1.0)
|
||||
if role == "scout":
|
||||
return 1.0 if params <= 40 else 0.5
|
||||
return 0.5
|
||||
|
||||
|
||||
def _catalog_role_match(role: str, name: str) -> bool:
|
||||
"""Ist dieses Modell im kuratierten Katalog (Cookbook) genau für DIESE Rolle gelistet?
|
||||
Dann ist es der prinzipien-konforme Pick → starker Bonus."""
|
||||
meta = catalog.meta_for_name(name)
|
||||
return bool(meta and (meta.get("role") or "").lower() == (role or "").lower())
|
||||
|
||||
|
||||
def _reason(role: str, caps: dict, name: str, fit: dict, fits: bool,
|
||||
incomplete: bool, cat_match: bool) -> str:
|
||||
if incomplete:
|
||||
return "Download unvollständig"
|
||||
if role == "vision" and not caps.get("vision"):
|
||||
return "keine Vision-Fähigkeit"
|
||||
if role == "coder" and not caps.get("coder"):
|
||||
return "kein Coder-Modell"
|
||||
if role == "hermes" and "hermes" not in (name or "").lower() and caps.get("tools") == "no":
|
||||
return "kein natives Tool-Calling"
|
||||
if not fits:
|
||||
return "passt nicht ins Budget (OOM)"
|
||||
bits = []
|
||||
if cat_match:
|
||||
bits.append("Katalog-Pick ✓")
|
||||
if role == "vision":
|
||||
bits.append("Vision ✓")
|
||||
if role == "coder" and caps.get("coder"):
|
||||
bits.append("Coder ✓")
|
||||
if role == "hermes":
|
||||
bits.append("Hermes" if "hermes" in (name or "").lower()
|
||||
else ("Tools ✓" if caps.get("tools") != "no" else "ohne Tools"))
|
||||
if caps.get("moe"):
|
||||
bits.append("MoE")
|
||||
bits.append(f"{fit['text']}, ~{fit['tps']:.0f} t/s")
|
||||
return " · ".join(bits)
|
||||
|
||||
|
||||
def recommend_for_role(role: str) -> dict:
|
||||
"""Rankt alle installierten Modelle für eine Rolle. Empfohlen = bester geeigneter,
|
||||
passender Eintrag. Liefert pro Modell Fit/Eignung/Begründung fürs UI."""
|
||||
role = (role or "").strip().lower()
|
||||
ram = _ram_gb()
|
||||
out = []
|
||||
for m in llamaswap.list_models():
|
||||
caps = m.get("capabilities") or {}
|
||||
params = budget.params_of_model(m)
|
||||
quant = m.get("quant") or "Q4_K_M"
|
||||
ctx = budget.setup_aware_ctx_for_model(m)["ctx"]
|
||||
fit = evaluate_fit(params, quant, ctx, ram, name=m["name"])
|
||||
incomplete = bool(m.get("incomplete"))
|
||||
fits = (fit["level"] != "too_tight") and not incomplete
|
||||
tps = fit["tps"] or 0
|
||||
suit = _capability_suit(role, caps, m["name"])
|
||||
cat_match = _catalog_role_match(role, m["name"])
|
||||
suitable = suit >= 0.5 and fits
|
||||
|
||||
fit_term = {"perfect": 1.0, "marginal": 0.3}.get(fit["level"], -2.0)
|
||||
# Eignung dominiert (×2), rollengerechte Größe/Tempo (_pref), Katalog-Anker, dann Fit.
|
||||
score = (2.0 * suit) + _pref(role, params, tps) + (0.6 if cat_match else 0.0) + fit_term
|
||||
if not fits:
|
||||
score -= 5.0
|
||||
|
||||
out.append({
|
||||
"name": m["name"], "current_role": m.get("role"),
|
||||
"params_b": round(params, 1), "quant": quant,
|
||||
"fit": fit, "suitable": suitable, "incomplete": incomplete,
|
||||
"score": round(score, 3),
|
||||
"reason": _reason(role, caps, m["name"], fit, fits, incomplete, cat_match),
|
||||
})
|
||||
|
||||
out.sort(key=lambda x: -x["score"])
|
||||
rec = next((o["name"] for o in out if o["suitable"]), None)
|
||||
for o in out:
|
||||
o["recommended"] = (o["name"] == rec)
|
||||
return {"role": role, "recommended": rec, "models": out}
|
||||
@@ -0,0 +1,90 @@
|
||||
"""
|
||||
Lane-Routing für den eingebauten MC2-Gateway (:9001/v1).
|
||||
|
||||
Zwei virtuelle Lanes, die Clients/IDEs auswählen — der Router pickt das echte Modell:
|
||||
- **chat** (= altes `auto`): Alltag → `fast`, schwer/lang → `heavy`.
|
||||
- **coding**: Code-Arbeit → `coder` (Qwen3-Coder-Next); riesiger/architektonischer Kontext → `heavy`;
|
||||
triviale Kurzfrage ohne Code → `fast` (Tempo).
|
||||
|
||||
Regelbasiert, sub-ms, ohne Cloud. Schwellen/Aliases liegen in einer UI-editierbaren Policy
|
||||
(routing_policy.py, hot-reload; Env = Defaults). Lucy läuft NICHT hierüber — die ist der
|
||||
Hermes-Agent (:8642), eigene Ebene.
|
||||
"""
|
||||
|
||||
import re
|
||||
|
||||
from services.routing_policy import load_policy
|
||||
|
||||
# Modell-Aliases & Zeichen-Schwellen liegen jetzt in der UI-editierbaren Policy
|
||||
# (routing_policy.py) und kommen pro Request via load_policy() (hot-reload). Die Env-Vars
|
||||
# sind dort die Defaults. Die Regex-Keyword-Listen unten bleiben bewusst im Code.
|
||||
|
||||
# Virtuelle Lanes, die im Gateway als „Modelle" sichtbar sind.
|
||||
LANES = ["coding", "chat"]
|
||||
LANE_ALIASES = {"auto": "chat"} # Rückwärtskompatibel: model:auto == chat
|
||||
|
||||
_HEAVY_KW = re.compile(
|
||||
r"\b(beweis|prove|theorem|komplex|complex|schwierig|"
|
||||
r"think\s*hard|reason\s*carefully|tief\s*nachdenk|optimi[sz]e|"
|
||||
r"root\s*cause|analy[sz]e\s+deeply|step[-\s]?by[-\s]?step)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
# Coding-spezifische „das ist groß/architektonisch" Signale → heavy statt coder.
|
||||
_CODING_HEAVY_KW = re.compile(
|
||||
r"\b(architekt|architect|system[-\s]?design|refactor\s+the\s+(whole|entire)|"
|
||||
r"ganze[ns]?\s+(architektur|codebase|projekt)|migrat\w+\s+(the\s+)?(whole|entire|gesamte)|"
|
||||
r"entwirf\s+(eine\s+)?architektur|plane?\s+(die\s+)?architektur)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
# Code-Indikatoren — verhindert, dass echte Code-Anfragen als „trivial" auf fast abrutschen.
|
||||
# Bewusst breit (inkl. natürlichsprachiger Coding-Begriffe DE+EN): in der coding-Lane soll im Zweifel
|
||||
# `coder` gewinnen; nur echte Nicht-Code-Kürze ("hallo", "wie spät") rutscht auf fast.
|
||||
_CODE_HINT = re.compile(
|
||||
r"```|\bdef \b|\bclass \b|\bimport \b|\bfunction\b|=>|;\s*$|"
|
||||
r"\.(py|ts|tsx|js|jsx|go|rs|java|cpp|c|rb|php|sql)\b|/src/|traceback|stack\s*trace|"
|
||||
r"\b(funktion|function|bug|fix|fehler|error|exception|implementier\w*|schreib\w*|"
|
||||
r"code\w*|coden|test\w*|klasse|method\w*|methode|refactor\w*|kompil\w*|compile|"
|
||||
r"build|deploy|debug|script|skript|api|endpoint|query|regex|json|yaml|"
|
||||
r"npm|pip|git|docker|terminal|shell|command)\b",
|
||||
re.IGNORECASE | re.MULTILINE,
|
||||
)
|
||||
|
||||
|
||||
def _text_of(body: dict) -> str:
|
||||
msgs = body.get("messages") or []
|
||||
return "\n".join(str(m.get("content") or "") for m in msgs)
|
||||
|
||||
|
||||
def _route_chat(text: str, n: int) -> tuple[str, str]:
|
||||
p = load_policy()
|
||||
if n > p["heavy_chars"]:
|
||||
return p["heavy"], f"langer Kontext ({n} > {p['heavy_chars']} Zeichen)"
|
||||
if _HEAVY_KW.search(text):
|
||||
return p["heavy"], "Komplexitäts-Schlüsselwort erkannt"
|
||||
return p["fast"], "Standard"
|
||||
|
||||
|
||||
def _route_coding(text: str, n: int) -> tuple[str, str]:
|
||||
# Agentisches Coden (OpenCode/RooCode/…) bleibt IMMER beim dedizierten Coder — NIE heavy/fast
|
||||
# (das sind Allzweck-Modelle, schwächer bei Code). Die Qwen-Coder packen 256K–1M Kontext selbst,
|
||||
# langer Repo-Kontext ist bei Agenten der Normalfall und darf NICHT zu heavy umrouten.
|
||||
# (Phase 2b: warme schnelle Coder-Stufe coder_lite als Default + coder als Eskalation.)
|
||||
p = load_policy()
|
||||
if p["coder_lite"] and not _CODING_HEAVY_KW.search(text) and n <= p["coding_escalate_chars"]:
|
||||
return p["coder_lite"], "Coding (schneller Coder)"
|
||||
return p["coder"], "Coding -> starker Coder"
|
||||
|
||||
|
||||
def choose_for_lane(lane: str, body: dict) -> tuple[str, str]:
|
||||
"""Wählt das echte Modell-Alias für eine Lane. Gibt (alias, begründung) zurück."""
|
||||
lane = LANE_ALIASES.get((lane or "chat").lower(), (lane or "chat").lower())
|
||||
text = _text_of(body)
|
||||
n = len(text)
|
||||
if lane == "coding":
|
||||
return _route_coding(text, n)
|
||||
return _route_chat(text, n) # chat + alles Unbekannte
|
||||
|
||||
|
||||
def choose_model(body: dict) -> tuple[str, str]:
|
||||
"""Rückwärtskompatibel: altes `model:auto` == chat-Lane."""
|
||||
return choose_for_lane("chat", body)
|
||||
@@ -0,0 +1,127 @@
|
||||
"""
|
||||
UI-editierbare Routing-Policy für die Gateway-Lanes (coding/chat).
|
||||
|
||||
Persistiert als JSON unter MC_ROUTING_POLICY_PATH (Default MODELS_DIR/mc2-routing.json —
|
||||
gleiche Konvention wie mc2-discover.json). **Hot-reload:** load_policy() liest die Datei nur
|
||||
bei Änderung neu (mtime-Cache) → UI-Edits greifen ohne Dienst-Neustart. Die Env-Vars (bisher
|
||||
einzige Stellschraube in router_logic.py) bleiben als Defaults/Fallback erhalten.
|
||||
|
||||
Bewusst NICHT editierbar (v1): die Regex-Keyword-Listen (heavy/coding-heavy/code-hint) — die
|
||||
bleiben in router_logic.py im Code.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import threading
|
||||
from pathlib import Path
|
||||
|
||||
from config import MODELS_DIR
|
||||
|
||||
POLICY_PATH = Path(os.environ.get("MC_ROUTING_POLICY_PATH", str(MODELS_DIR / "mc2-routing.json")))
|
||||
|
||||
|
||||
def _env_bool(name: str, default: str) -> bool:
|
||||
return os.environ.get(name, default) not in ("0", "false", "")
|
||||
|
||||
|
||||
# Defaults aus den Env-Vars — Quelle der Wahrheit, solange keine Policy-Datei existiert.
|
||||
DEFAULTS: dict = {
|
||||
"fast": os.environ.get("MC_ROUTE_FAST", "fast"),
|
||||
"heavy": os.environ.get("MC_ROUTE_HEAVY", "heavy"),
|
||||
"coder": os.environ.get("MC_ROUTE_CODER", "coder"),
|
||||
"coder_lite": os.environ.get("MC_ROUTE_CODER_LITE", "").strip(),
|
||||
"heavy_chars": int(os.environ.get("MC_GATEWAY_HEAVY_CHARS", "8000")),
|
||||
"coding_escalate_chars": int(os.environ.get("MC_CODING_ESCALATE_CHARS", "120000")),
|
||||
"fast_no_think": _env_bool("MC_FAST_NO_THINK", "1"),
|
||||
}
|
||||
|
||||
# Feld-Spezifikation für die UI (Typ + Grenzen + Label). Treibt Editor & Validierung.
|
||||
FIELDS: list[dict] = [
|
||||
{"key": "fast", "label": "fast-Alias (chat: Standard)", "type": "str"},
|
||||
{"key": "heavy", "label": "heavy-Alias (chat: lang/komplex)", "type": "str"},
|
||||
{"key": "coder", "label": "coder-Alias (coding: stark / Eskalation)", "type": "str"},
|
||||
{"key": "coder_lite", "label": "coder-lite-Alias (coding: schneller Default; leer = aus)", "type": "str"},
|
||||
{"key": "heavy_chars", "label": "chat → heavy ab N Zeichen", "type": "int", "min": 500, "max": 1_000_000},
|
||||
{"key": "coding_escalate_chars", "label": "coding → starker Coder ab N Zeichen", "type": "int", "min": 1000, "max": 4_000_000},
|
||||
{"key": "fast_no_think", "label": "fast-Spur: Thinking aus (flotte Antworten)", "type": "bool"},
|
||||
]
|
||||
|
||||
_LOCK = threading.Lock()
|
||||
_CACHE: dict = {"mtime": None, "policy": None}
|
||||
|
||||
|
||||
def _read_file() -> dict:
|
||||
try:
|
||||
with open(POLICY_PATH, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
return data if isinstance(data, dict) else {}
|
||||
except (FileNotFoundError, json.JSONDecodeError, OSError):
|
||||
return {}
|
||||
|
||||
|
||||
def _coerce(patch: dict) -> dict:
|
||||
"""Nur bekannte Keys, typ-/bereichsvalidiert. Wirft ValueError bei ungültigen Werten."""
|
||||
spec = {f["key"]: f for f in FIELDS}
|
||||
out: dict = {}
|
||||
for k, v in (patch or {}).items():
|
||||
f = spec.get(k)
|
||||
if not f:
|
||||
continue # unbekannte Keys still verwerfen
|
||||
if f["type"] == "int":
|
||||
iv = int(v)
|
||||
lo, hi = f.get("min", 1), f.get("max", 10**9)
|
||||
if not (lo <= iv <= hi):
|
||||
raise ValueError(f"{k}={iv} außerhalb [{lo}, {hi}]")
|
||||
out[k] = iv
|
||||
elif f["type"] == "bool":
|
||||
out[k] = bool(v)
|
||||
else: # str
|
||||
sv = str(v).strip()
|
||||
if k != "coder_lite" and not sv:
|
||||
raise ValueError(f"{k} darf nicht leer sein")
|
||||
out[k] = sv
|
||||
return out
|
||||
|
||||
|
||||
def _coerce_safe(patch: dict) -> dict:
|
||||
"""Wie _coerce, aber schluckt Fehler — kaputte Datei darf den Betrieb nicht stoppen."""
|
||||
try:
|
||||
return _coerce(patch)
|
||||
except (ValueError, TypeError):
|
||||
return {}
|
||||
|
||||
|
||||
def load_policy() -> dict:
|
||||
"""Aktuelle Policy (Datei über DEFAULTS gemerged). Hot-reload via mtime-Cache, pro Request billig."""
|
||||
try:
|
||||
mtime = POLICY_PATH.stat().st_mtime
|
||||
except OSError:
|
||||
mtime = None
|
||||
with _LOCK:
|
||||
if _CACHE["policy"] is None or _CACHE["mtime"] != mtime:
|
||||
merged = {**DEFAULTS}
|
||||
if mtime is not None:
|
||||
merged.update(_coerce_safe(_read_file()))
|
||||
_CACHE["mtime"] = mtime
|
||||
_CACHE["policy"] = merged
|
||||
return dict(_CACHE["policy"])
|
||||
|
||||
|
||||
def save_policy(patch: dict) -> dict:
|
||||
"""Validiert + persistiert atomar. Gibt die neue, vollständige Policy zurück."""
|
||||
clean = _coerce(patch) # wirft bei ungültigem Input
|
||||
with _LOCK:
|
||||
current = {**DEFAULTS, **_coerce_safe(_read_file()), **clean}
|
||||
POLICY_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = POLICY_PATH.with_suffix(".json.tmp")
|
||||
with open(tmp, "w", encoding="utf-8") as f:
|
||||
json.dump(current, f, ensure_ascii=False, indent=2)
|
||||
os.replace(tmp, POLICY_PATH)
|
||||
_CACHE["mtime"] = None # nächster load_policy() lädt frisch
|
||||
_CACHE["policy"] = None
|
||||
return current
|
||||
|
||||
|
||||
def policy_meta() -> dict:
|
||||
"""Für den UI-Editor: aktuelle Werte + Defaults (für „Zurücksetzen“) + Feld-Spezifikation."""
|
||||
return {"policy": load_policy(), "defaults": dict(DEFAULTS), "fields": FIELDS}
|
||||
@@ -0,0 +1,35 @@
|
||||
"""
|
||||
Vertrauenswürdige Quellen + Kategorien für die automatische Modell-Entdeckung.
|
||||
Rollen = die EINE Quelle der Wahrheit, identisch zu den llama-swap-Serving-Rollen
|
||||
und der UI: fast · heavy · coder · vision · hermes (Agent-Hirn) · scout.
|
||||
"""
|
||||
|
||||
# HF-Orgs, die zuverlässig aktuelle, hochwertige GGUF-Quants veröffentlichen.
|
||||
TRUSTED_AUTHORS = ["unsloth", "bartowski", "ggml-org", "lmstudio-community"]
|
||||
|
||||
# Kanonische Rollen — eine Quelle der Wahrheit (deckt sich mit llamaswap.ROLE_IDS,
|
||||
# maintenance.ROLE_MAP, frontend ModelBadges.ROLES + Discover.ROLE_METADATA).
|
||||
# `hermes` = Lucys Agent-Hirn (warm + ko-resident); UI-Label „Hirn".
|
||||
ROLE_IDS = ["fast", "heavy", "coder", "vision", "hermes", "scout"]
|
||||
|
||||
# Kategorien (Reihenfolge = Anzeige + Zuordnungs-Priorität). Ein Modell wird der
|
||||
# ERSTEN Kategorie zugeordnet, deren Stichwort im Repo-Namen vorkommt; sonst „scout".
|
||||
# Die `role` ist zugleich der Alias-Vorschlag und EINE der 5 kanonischen Rollen.
|
||||
CATEGORIES = [
|
||||
{"role": "vision", "title": "Bilder verstehen", "icon": "eye",
|
||||
"kw": ["-vl-", "-vl", "vision", "llava", "multimodal", "-mm-", "pixtral"]},
|
||||
{"role": "coder", "title": "Coden & Programmieren", "icon": "code",
|
||||
"kw": ["coder", "-code-", "code-", "codestral", "starcoder"]},
|
||||
{"role": "hermes", "title": "Lucys Hirn (Agent)", "icon": "brain-circuit",
|
||||
"kw": ["hermes"]}, # Agent-Hirn: Hermes-Familie am robustesten (natives Tool-Calling).
|
||||
{"role": "heavy", "title": "Schweres Reasoning", "icon": "brain",
|
||||
"kw": ["reasoning", "-think", "thinking", "gpt-oss", "deepseek-r", "-r1", "qwq",
|
||||
"-70b", "-72b", "-120b", "-123b", "-235b", "-405b", "-a10b", "-a22b"]},
|
||||
{"role": "fast", "title": "Schnelles Alltags-Hirn", "icon": "zap",
|
||||
"kw": ["-a3b", "-a1", "-a2", "-30b", "-32b", "-14b", "-8b", "-7b", "-4b", "-moe"]},
|
||||
{"role": "scout", "title": "Multimodal-Allrounder", "icon": "compass",
|
||||
"kw": []}, # Fallback: instruct/chat-Modelle, die in keine Spezialrolle fallen
|
||||
]
|
||||
|
||||
# Repo-Namensteile, die bei der Entdeckung übersprungen werden (Roh-/Spezialformate).
|
||||
SKIP_TOKENS = ["-base", "-bnb-", "-gptq", "-awq", "-fp8", "draft", "tokenizer"]
|
||||
@@ -0,0 +1,195 @@
|
||||
"""
|
||||
System/OS-Metriken für die Box (Bosgame / Strix Halo).
|
||||
|
||||
CPU/RAM/Disk via psutil (plattformübergreifend). GPU-Auslastung/VRAM/Temperatur
|
||||
via sysfs (amdgpu) — nur Linux; auf anderen Plattformen None (amd-smi fehlt auf
|
||||
der Box, daher sysfs). Verschachtelte Struktur wie v1 (cpu.percent, ram.used Bytes).
|
||||
"""
|
||||
|
||||
import glob
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
import psutil
|
||||
|
||||
from config import MODELS_DIR
|
||||
|
||||
|
||||
def _read_int(path: str) -> int | None:
|
||||
try:
|
||||
with open(path) as f:
|
||||
return int(f.read().strip())
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _gpu_sysfs() -> dict | None:
|
||||
"""AMD-GPU-Auslastung + Speicher via sysfs (Linux). Findet die Basis-Card
|
||||
dynamisch (Strix Halo ist oft card1, nicht card0) und überspringt die
|
||||
Connector-Verzeichnisse (card1-DP-1 …). Strix Halo nutzt Unified Memory →
|
||||
GTT ist der eigentliche große Pool; VRAM ist nur der kleine Carve-out."""
|
||||
for dev in sorted(glob.glob("/sys/class/drm/card*/device")):
|
||||
card = dev.split("/")[-2] # z.B. "card1" oder "card1-DP-1"
|
||||
if "-" in card: # Connector-Dir → kein GPU-Device
|
||||
continue
|
||||
busy = _read_int(f"{dev}/gpu_busy_percent")
|
||||
if busy is None:
|
||||
continue
|
||||
return {
|
||||
"busy_percent": busy,
|
||||
"vram_used": _read_int(f"{dev}/mem_info_vram_used"),
|
||||
"vram_total": _read_int(f"{dev}/mem_info_vram_total"),
|
||||
"gtt_used": _read_int(f"{dev}/mem_info_gtt_used"),
|
||||
"gtt_total": _read_int(f"{dev}/mem_info_gtt_total"),
|
||||
}
|
||||
return None
|
||||
|
||||
|
||||
def _temps() -> dict | None:
|
||||
"""CPU/GPU-Temperatur via hwmon (Linux). None bei Fehlen."""
|
||||
out: dict = {}
|
||||
for hw in glob.glob("/sys/class/hwmon/hwmon*"):
|
||||
name = ""
|
||||
try:
|
||||
with open(f"{hw}/name") as f:
|
||||
name = f.read().strip()
|
||||
except Exception:
|
||||
continue
|
||||
t = _read_int(f"{hw}/temp1_input")
|
||||
if t is None:
|
||||
continue
|
||||
c = round(t / 1000.0, 1)
|
||||
if name in ("k10temp", "zenpower", "coretemp"):
|
||||
out["cpu"] = c
|
||||
elif name in ("amdgpu", "edge"):
|
||||
out["gpu"] = c
|
||||
return out or None
|
||||
|
||||
|
||||
def get_git_info(path: str) -> dict | None:
|
||||
expanded = os.path.expanduser(path)
|
||||
if not os.path.isdir(expanded) or not os.path.exists(os.path.join(expanded, ".git")):
|
||||
return None
|
||||
try:
|
||||
res = subprocess.run(
|
||||
["git", "log", "-1", "--format=%h|%cd|%s", "--date=short"],
|
||||
cwd=expanded, capture_output=True, text=True, timeout=3
|
||||
)
|
||||
if res.returncode != 0:
|
||||
return None
|
||||
parts = res.stdout.strip().split("|", 2)
|
||||
h = parts[0]
|
||||
d = parts[1]
|
||||
s = parts[2] if len(parts) > 2 else ""
|
||||
|
||||
branch_res = subprocess.run(
|
||||
["git", "rev-parse", "--abbrev-ref", "HEAD"],
|
||||
cwd=expanded, capture_output=True, text=True, timeout=2
|
||||
)
|
||||
branch = branch_res.stdout.strip() if branch_res.returncode == 0 else "unknown"
|
||||
|
||||
status_res = subprocess.run(
|
||||
["git", "status", "--porcelain"],
|
||||
cwd=expanded, capture_output=True, text=True, timeout=2
|
||||
)
|
||||
dirty = bool(status_res.stdout.strip()) if status_res.returncode == 0 else False
|
||||
|
||||
return {
|
||||
"hash": h,
|
||||
"date": d,
|
||||
"subject": s,
|
||||
"branch": branch,
|
||||
"dirty": dirty,
|
||||
"path": expanded
|
||||
}
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def find_hermes_agent_git() -> dict | None:
|
||||
env_path = os.environ.get("MC_HERMES_AGENT_PATH")
|
||||
if env_path:
|
||||
info = get_git_info(env_path)
|
||||
if info:
|
||||
return info
|
||||
|
||||
candidates = [
|
||||
"~/hermes-agent",
|
||||
"~/.hermes/hermes-agent",
|
||||
"~/hermes-webui/hermes-agent",
|
||||
"~/.hermes"
|
||||
]
|
||||
for c in candidates:
|
||||
info = get_git_info(c)
|
||||
if info:
|
||||
return info
|
||||
return None
|
||||
|
||||
|
||||
def get_engine_version() -> dict:
|
||||
engine_path = os.environ.get("MC_ENGINE_PATH", "/opt/llamacpp")
|
||||
git_info = get_git_info(engine_path)
|
||||
if git_info:
|
||||
return {**git_info, "type": "git"}
|
||||
|
||||
candidates = [
|
||||
os.path.join(engine_path, "llama-server"),
|
||||
os.path.join(engine_path, "bin", "llama-server"),
|
||||
"/usr/local/bin/llama-server",
|
||||
"/usr/bin/llama-server",
|
||||
"llama-server"
|
||||
]
|
||||
|
||||
for binary in candidates:
|
||||
if binary != "llama-server" and not os.path.exists(binary):
|
||||
continue
|
||||
try:
|
||||
res = subprocess.run([binary, "--version"], capture_output=True, text=True, timeout=2)
|
||||
output = (res.stdout or "").strip() or (res.stderr or "").strip()
|
||||
if output:
|
||||
lines = output.splitlines()
|
||||
ver = lines[0] if lines else "unknown"
|
||||
return {"version_text": ver, "type": "binary"}
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return {"type": "unknown"}
|
||||
|
||||
|
||||
_VERSION_CACHE = {"ts": 0.0, "data": {}}
|
||||
|
||||
|
||||
def check_versions_cached() -> dict:
|
||||
import time
|
||||
now = time.time()
|
||||
if now - _VERSION_CACHE["ts"] < 30.0:
|
||||
return _VERSION_CACHE["data"]
|
||||
|
||||
mc2_path = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", ".."))
|
||||
|
||||
data = {
|
||||
"mc2": get_git_info(mc2_path),
|
||||
"engine": get_engine_version(),
|
||||
"hermes_ui": get_git_info("~/hermes-webui"),
|
||||
"hermes_agent": find_hermes_agent_git()
|
||||
}
|
||||
_VERSION_CACHE["ts"] = now
|
||||
_VERSION_CACHE["data"] = data
|
||||
return data
|
||||
|
||||
|
||||
def system_status() -> dict:
|
||||
vm = psutil.virtual_memory()
|
||||
try:
|
||||
du = psutil.disk_usage(str(MODELS_DIR) if MODELS_DIR.exists() else os.getcwd())
|
||||
disk = {"total": du.total, "used": du.used, "percent": du.percent}
|
||||
except Exception:
|
||||
disk = None
|
||||
return {
|
||||
"cpu": {"percent": psutil.cpu_percent(interval=0.1), "cores": psutil.cpu_count()},
|
||||
"ram": {"total": vm.total, "used": vm.used, "percent": vm.percent},
|
||||
"gpu": _gpu_sysfs(),
|
||||
"temp": _temps(),
|
||||
"disk": disk,
|
||||
"versions": check_versions_cached(),
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
"""Token-Statistik (Verbrauch je Modell) mit gedrosseltem Persistieren.
|
||||
|
||||
Früher wurde bei JEDEM Request die komplette JSON-Datei gelesen und geschrieben
|
||||
(Disk-Thrash). Jetzt: einmaliges Laden in einen In-Memory-Cache, Inkremente laufen
|
||||
gegen den Cache, Persistieren passiert höchstens alle FLUSH_INTERVAL Sekunden sowie
|
||||
beim Prozess-Ende (atexit). Lesen liefert immer den aktuellen (auch ungeflushten) Stand.
|
||||
"""
|
||||
|
||||
import atexit
|
||||
import json
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
from config import HERMES_HOME
|
||||
|
||||
STATS_FILE = HERMES_HOME / "token_stats.json"
|
||||
FLUSH_INTERVAL = 5.0 # Sekunden zwischen Disk-Writes
|
||||
# Baseline (repräsentiert Verbrauch vor dem modellspezifischen Logging).
|
||||
_BASELINE = {"prompt_tokens": 718400, "completion_tokens": 324200, "models": {}}
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_lock = threading.Lock()
|
||||
_stats: dict | None = None
|
||||
_dirty = False
|
||||
_last_flush = 0.0
|
||||
|
||||
|
||||
def _load_from_disk() -> dict:
|
||||
if not STATS_FILE.exists():
|
||||
return dict(_BASELINE)
|
||||
try:
|
||||
with open(STATS_FILE, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
data.setdefault("prompt_tokens", 0)
|
||||
data.setdefault("completion_tokens", 0)
|
||||
data.setdefault("models", {})
|
||||
return data
|
||||
except (OSError, json.JSONDecodeError):
|
||||
log.warning("token_stats: Laden fehlgeschlagen, nutze Baseline", exc_info=True)
|
||||
return dict(_BASELINE)
|
||||
|
||||
|
||||
def _ensure_loaded() -> dict:
|
||||
global _stats
|
||||
if _stats is None:
|
||||
_stats = _load_from_disk()
|
||||
return _stats
|
||||
|
||||
|
||||
def _write(stats: dict) -> None:
|
||||
try:
|
||||
STATS_FILE.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = STATS_FILE.with_suffix(".tmp")
|
||||
with open(tmp, "w", encoding="utf-8") as f:
|
||||
json.dump(stats, f)
|
||||
tmp.replace(STATS_FILE)
|
||||
except OSError:
|
||||
log.warning("token_stats: Schreiben fehlgeschlagen", exc_info=True)
|
||||
|
||||
|
||||
def get_stats() -> dict:
|
||||
"""Aktueller Stand (inkl. noch nicht geflushter Inkremente) als Kopie."""
|
||||
with _lock:
|
||||
return json.loads(json.dumps(_ensure_loaded()))
|
||||
|
||||
|
||||
def increment_tokens(prompt: int, completion: int, model: str | None = None) -> None:
|
||||
"""Tokens im Cache verbuchen; gedrosselt auf Disk persistieren."""
|
||||
global _dirty, _last_flush
|
||||
with _lock:
|
||||
stats = _ensure_loaded()
|
||||
stats["prompt_tokens"] += prompt
|
||||
stats["completion_tokens"] += completion
|
||||
if model:
|
||||
m = stats.setdefault("models", {}).setdefault(
|
||||
model.lower(), {"prompt": 0, "completion": 0})
|
||||
m["prompt"] += prompt
|
||||
m["completion"] += completion
|
||||
_dirty = True
|
||||
now = time.monotonic()
|
||||
if now - _last_flush >= FLUSH_INTERVAL:
|
||||
_write(stats)
|
||||
_dirty = False
|
||||
_last_flush = now
|
||||
|
||||
|
||||
def flush() -> None:
|
||||
"""Ungeschriebene Inkremente sofort persistieren (z.B. beim Shutdown)."""
|
||||
global _dirty
|
||||
with _lock:
|
||||
if _dirty and _stats is not None:
|
||||
_write(_stats)
|
||||
_dirty = False
|
||||
|
||||
|
||||
atexit.register(flush)
|
||||
@@ -0,0 +1,68 @@
|
||||
"""
|
||||
Per-Stage-Latenz-Metriken für die Voice/Lucy-Pipeline.
|
||||
|
||||
Misst die Server-seitige Dauer jeder Stufe (STT, Vision-Beschreibung, Chat-TTFB, TTS) und hält
|
||||
rollende Statistiken (avg/p50/p95/last) im Speicher. Macht aus Latenz-VERMUTUNGEN gemessene Fakten
|
||||
— die eigentliche Voraussetzung, um gezielt zu optimieren (Stufe 5/C2 des Reviews). Anzeige im
|
||||
Frontend-Overhaul (E) analog zur TokenPerformanceCard.
|
||||
|
||||
In-Memory + thread-safe (keine Datei-I/O — Latenz-Telemetrie ist transient, Restart = Reset).
|
||||
"""
|
||||
|
||||
import threading
|
||||
import time
|
||||
from collections import deque
|
||||
|
||||
_LOCK = threading.Lock()
|
||||
_MAX = 200
|
||||
_STAGES: dict[str, deque] = {}
|
||||
|
||||
# Bekannte Stufen (für stabile UI-Reihenfolge); unbekannte werden trotzdem erfasst.
|
||||
STAGES = ("stt", "vision", "chat_ttfb", "tts")
|
||||
|
||||
|
||||
def record_stage(stage: str, ms: float) -> None:
|
||||
"""Eine gemessene Stage-Dauer (ms) verbuchen. No-op bei negativen Werten."""
|
||||
if ms is None or ms < 0:
|
||||
return
|
||||
with _LOCK:
|
||||
dq = _STAGES.get(stage)
|
||||
if dq is None:
|
||||
dq = _STAGES[stage] = deque(maxlen=_MAX)
|
||||
dq.append(float(ms))
|
||||
|
||||
|
||||
class Timer:
|
||||
"""Context-Manager: misst die verstrichene Zeit und verbucht sie auf `stage`.
|
||||
Funktioniert um `await`-Aufrufe herum (enter → await → exit)."""
|
||||
|
||||
def __init__(self, stage: str) -> None:
|
||||
self.stage = stage
|
||||
self._t0 = 0.0
|
||||
|
||||
def __enter__(self) -> "Timer":
|
||||
self._t0 = time.perf_counter()
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc) -> None:
|
||||
record_stage(self.stage, (time.perf_counter() - self._t0) * 1000.0)
|
||||
|
||||
|
||||
def _summary(vals: list[float]) -> dict:
|
||||
if not vals:
|
||||
return {"count": 0}
|
||||
s = sorted(vals)
|
||||
n = len(s)
|
||||
return {
|
||||
"count": n,
|
||||
"avg_ms": round(sum(s) / n, 1),
|
||||
"p50_ms": round(s[n // 2], 1),
|
||||
"p95_ms": round(s[min(n - 1, int(n * 0.95))], 1),
|
||||
"last_ms": round(vals[-1], 1),
|
||||
}
|
||||
|
||||
|
||||
def get_metrics() -> dict:
|
||||
"""Rollende Zusammenfassung je Stufe."""
|
||||
with _LOCK:
|
||||
return {stage: _summary(list(dq)) for stage, dq in _STAGES.items()}
|
||||
@@ -0,0 +1,77 @@
|
||||
"""
|
||||
Hält das Agent-Hirn (Rolle `hermes`) dauerhaft warm.
|
||||
|
||||
Hintergrund: llama-swap ist EIN-Gruppen-resident — lädt ein on-demand-Modell außerhalb
|
||||
der `brains`-Gruppe, wird die ganze Gruppe (inkl. Hirn) verdrängt. `persist: true`
|
||||
verhindert nur Idle-Unload, NICHT die Gruppen-Verdrängung; neu vorgewärmt wird sonst erst
|
||||
beim nächsten llama-swap-(Re)Start. Dieser Wächter schließt die Lücke: ist die Box idle
|
||||
(nichts geladen), pingt er das Hirn vor. Während aktiver Last (irgendetwas geladen) hält
|
||||
er sich raus, verdrängt also nie ein gerade genutztes Modell.
|
||||
|
||||
Abschaltbar/justierbar via Env: MC_REWARM_ENABLED=0, MC_REWARM_INTERVAL, MC_REWARM_MODEL.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
|
||||
import httpx
|
||||
|
||||
from config import LLAMA_SWAP_URL
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
ENABLED = os.environ.get("MC_REWARM_ENABLED", "1") != "0"
|
||||
INTERVAL = int(os.environ.get("MC_REWARM_INTERVAL", "90")) # Sekunden zwischen Checks
|
||||
START_DELAY = int(os.environ.get("MC_REWARM_START_DELAY", "25"))
|
||||
|
||||
|
||||
def _brain_model() -> str:
|
||||
"""Aktives Agent-Hirn = Hermes' model.default (sonst model.model). So wärmt der Wächter
|
||||
immer das WIRKLICH genutzte Hirn (passt sich Hirn-Wechseln an). Override: MC_REWARM_MODEL."""
|
||||
forced = os.environ.get("MC_REWARM_MODEL")
|
||||
if forced:
|
||||
return forced
|
||||
try:
|
||||
from ruamel.yaml import YAML
|
||||
from config import HERMES_HOME
|
||||
p = HERMES_HOME / "config.yaml"
|
||||
if p.exists():
|
||||
with p.open(encoding="utf-8") as f:
|
||||
cfg = YAML().load(f) or {}
|
||||
m = (cfg.get("model") or {}) if isinstance(cfg, dict) else {}
|
||||
v = m.get("default") or m.get("model")
|
||||
if v:
|
||||
return str(v)
|
||||
except Exception:
|
||||
log.debug("rewarm: Hirn-Lookup fehlgeschlagen", exc_info=True)
|
||||
return "fast"
|
||||
|
||||
|
||||
async def _running_empty() -> bool:
|
||||
async with httpx.AsyncClient(timeout=8.0) as c:
|
||||
r = await c.get(f"{LLAMA_SWAP_URL}/running")
|
||||
data = r.json() or {}
|
||||
return not (data.get("running") or [])
|
||||
|
||||
|
||||
async def _warm(model: str) -> None:
|
||||
async with httpx.AsyncClient(timeout=180.0) as c:
|
||||
await c.post(f"{LLAMA_SWAP_URL}/v1/chat/completions", json={
|
||||
"model": model, "max_tokens": 1,
|
||||
"messages": [{"role": "user", "content": "ping"}],
|
||||
})
|
||||
|
||||
|
||||
async def rewarm_loop() -> None:
|
||||
"""Endlos-Schleife (Hintergrund-Task): prüft periodisch, wärmt bei Idle vor."""
|
||||
await asyncio.sleep(START_DELAY) # Box/Engine nach MC-Start setzen lassen
|
||||
while True:
|
||||
try:
|
||||
if await _running_empty():
|
||||
model = _brain_model()
|
||||
log.info("rewarm: Box idle → Hirn '%s' wird vorgewärmt", model)
|
||||
await _warm(model)
|
||||
except Exception:
|
||||
log.debug("rewarm: Tick fehlgeschlagen", exc_info=True)
|
||||
await asyncio.sleep(INTERVAL)
|
||||
@@ -0,0 +1,3 @@
|
||||
.venv/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
@@ -0,0 +1,171 @@
|
||||
"""
|
||||
Hermes PC Executor — läuft auf dem Windows PC.
|
||||
Empfängt Tool-Befehle vom MCP-Server der AI Box und führt sie lokal aus.
|
||||
"""
|
||||
import base64
|
||||
import hmac
|
||||
import io
|
||||
import os
|
||||
import socket
|
||||
import subprocess
|
||||
import webbrowser
|
||||
from urllib.parse import quote
|
||||
|
||||
import uvicorn
|
||||
from fastapi import Depends, FastAPI, Header, HTTPException
|
||||
|
||||
app = FastAPI(title="Hermes PC Executor")
|
||||
|
||||
MY_PORT = int(os.environ.get("HERMES_PC_PORT", "7777"))
|
||||
# Auf welchem Interface lauschen. Default 0.0.0.0 (LAN), per Env einschränkbar (z.B. die LAN-IP des PCs).
|
||||
MY_HOST = os.environ.get("HERMES_PC_HOST", "0.0.0.0")
|
||||
|
||||
# ── Auth ───────────────────────────────────────────────────────────────────
|
||||
# Shared Secret. OHNE Token sind die gefährlichen Endpunkte (Shell/Eingabe/Öffnen/Screenshot)
|
||||
# fail-closed gesperrt → ein un-konfigurierter Executor ist KEINE offene Remote-Code-Execution mehr.
|
||||
# Der Token muss identisch auf der Box (Hermes-Env PC_EXECUTOR_TOKEN → mcp_pc.py) gesetzt sein.
|
||||
AUTH_TOKEN = os.environ.get("HERMES_PC_TOKEN", "").strip()
|
||||
|
||||
|
||||
def require_auth(authorization: str | None = Header(default=None)) -> None:
|
||||
"""Bearer-Token-Prüfung (konstante Zeit). 503 wenn der Executor ohne Token läuft (fail-closed),
|
||||
401 bei fehlendem/falschem Token."""
|
||||
if not AUTH_TOKEN:
|
||||
raise HTTPException(
|
||||
503,
|
||||
"Executor ohne HERMES_PC_TOKEN gestartet — Steuer-Endpunkte sind aus Sicherheitsgründen "
|
||||
"gesperrt. Setze die Env-Variable HERMES_PC_TOKEN (identisch zur Box) und starte neu.",
|
||||
)
|
||||
expected = f"Bearer {AUTH_TOKEN}"
|
||||
if not authorization or not hmac.compare_digest(authorization, expected):
|
||||
raise HTTPException(401, "Ungültiges oder fehlendes Bearer-Token.")
|
||||
|
||||
|
||||
# ── Shell ──────────────────────────────────────────────────────────────────
|
||||
|
||||
@app.post("/shell", dependencies=[Depends(require_auth)])
|
||||
async def run_shell(req: dict):
|
||||
cmd = req.get("command", "")
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["powershell", "-NoProfile", "-NonInteractive", "-Command", cmd],
|
||||
capture_output=True, text=True, timeout=60,
|
||||
encoding="utf-8", errors="replace",
|
||||
)
|
||||
return {
|
||||
"stdout": result.stdout.strip(),
|
||||
"stderr": result.stderr.strip(),
|
||||
"returncode": result.returncode,
|
||||
}
|
||||
except subprocess.TimeoutExpired:
|
||||
raise HTTPException(408, "Timeout nach 60s")
|
||||
except Exception as e:
|
||||
raise HTTPException(500, str(e))
|
||||
|
||||
|
||||
# ── Screen ─────────────────────────────────────────────────────────────────
|
||||
|
||||
@app.post("/screenshot", dependencies=[Depends(require_auth)])
|
||||
async def take_screenshot():
|
||||
try:
|
||||
import mss
|
||||
from PIL import Image
|
||||
|
||||
with mss.mss() as sct:
|
||||
monitor = sct.monitors[1]
|
||||
raw = sct.grab(monitor)
|
||||
img = Image.frombytes("RGB", raw.size, raw.bgra, "raw", "BGRX")
|
||||
img.thumbnail((1280, 720))
|
||||
buf = io.BytesIO()
|
||||
img.save(buf, format="JPEG", quality=75)
|
||||
return {"screenshot": base64.b64encode(buf.getvalue()).decode()}
|
||||
except Exception as e:
|
||||
raise HTTPException(500, str(e))
|
||||
|
||||
|
||||
# ── Input ──────────────────────────────────────────────────────────────────
|
||||
|
||||
@app.post("/type", dependencies=[Depends(require_auth)])
|
||||
async def type_text(req: dict):
|
||||
text = req.get("text", "")
|
||||
try:
|
||||
import pyautogui
|
||||
pyautogui.write(text, interval=0.03)
|
||||
return {"ok": True}
|
||||
except Exception as e:
|
||||
raise HTTPException(500, str(e))
|
||||
|
||||
|
||||
@app.post("/key", dependencies=[Depends(require_auth)])
|
||||
async def press_keys(req: dict):
|
||||
keys = req.get("keys", "")
|
||||
try:
|
||||
import pyautogui
|
||||
parts = [k.strip() for k in keys.split("+")]
|
||||
pyautogui.hotkey(*parts)
|
||||
return {"ok": True}
|
||||
except Exception as e:
|
||||
raise HTTPException(500, str(e))
|
||||
|
||||
|
||||
# ── Apps / Browser ─────────────────────────────────────────────────────────
|
||||
|
||||
@app.post("/open", dependencies=[Depends(require_auth)])
|
||||
async def open_target(req: dict):
|
||||
target = req.get("target", "")
|
||||
try:
|
||||
if target.startswith("http://") or target.startswith("https://"):
|
||||
webbrowser.open(target)
|
||||
else:
|
||||
os.startfile(target)
|
||||
return {"ok": True}
|
||||
except Exception as e:
|
||||
raise HTTPException(500, str(e))
|
||||
|
||||
|
||||
@app.post("/search", dependencies=[Depends(require_auth)])
|
||||
async def search_web(req: dict):
|
||||
query = req.get("query", "")
|
||||
url = f"https://www.google.com/search?q={quote(query)}"
|
||||
webbrowser.open(url)
|
||||
return {"ok": True, "url": url}
|
||||
|
||||
|
||||
# ── Health ─────────────────────────────────────────────────────────────────
|
||||
|
||||
@app.get("/health")
|
||||
async def health():
|
||||
return {"status": "ok", "host": socket.gethostname()}
|
||||
|
||||
|
||||
def _get_local_ip() -> str:
|
||||
try:
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as s:
|
||||
s.connect(("8.8.8.8", 80))
|
||||
return s.getsockname()[0]
|
||||
except Exception:
|
||||
return socket.gethostbyname(socket.gethostname())
|
||||
|
||||
|
||||
def _ensure_streams() -> None:
|
||||
"""Unter pythonw (kein Konsolenfenster) sind sys.stdout/stderr = None →
|
||||
uvicorns Logging crasht beim Start. Dann auf eine Logdatei umbiegen."""
|
||||
import sys
|
||||
if sys.stdout is not None and sys.stderr is not None:
|
||||
return
|
||||
logdir = os.path.join(
|
||||
os.environ.get("LOCALAPPDATA", os.path.dirname(os.path.abspath(__file__))),
|
||||
"HermesPCExecutor",
|
||||
)
|
||||
os.makedirs(logdir, exist_ok=True)
|
||||
f = open(os.path.join(logdir, "executor.log"), "a", buffering=1, encoding="utf-8")
|
||||
sys.stdout = f
|
||||
sys.stderr = f
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
_ensure_streams()
|
||||
my_ip = _get_local_ip()
|
||||
auth_state = "Token AKTIV" if AUTH_TOKEN else "KEIN Token → Steuer-Endpunkte GESPERRT (fail-closed)"
|
||||
print(f"Hermes PC Executor läuft auf http://{my_ip}:{MY_PORT} (bind {MY_HOST}) — {auth_state}")
|
||||
uvicorn.run(app, host=MY_HOST, port=MY_PORT)
|
||||
@@ -0,0 +1,8 @@
|
||||
@echo off
|
||||
cd /d "%~dp0"
|
||||
echo === Hermes PC Executor Setup ===
|
||||
python -m venv .venv
|
||||
.venv\Scripts\pip install -r requirements.txt -q
|
||||
echo.
|
||||
echo Fertig. Starte mit start.bat
|
||||
pause
|
||||
@@ -0,0 +1,6 @@
|
||||
fastapi>=0.111.0
|
||||
uvicorn>=0.30.0
|
||||
httpx>=0.27.0
|
||||
mss>=9.0.1
|
||||
Pillow>=10.0.0
|
||||
pyautogui>=0.9.54
|
||||
@@ -0,0 +1,9 @@
|
||||
@echo off
|
||||
cd /d "%~dp0"
|
||||
echo === Hermes PC Executor ===
|
||||
echo Port: 7777
|
||||
echo Hermes (AI Box) kann jetzt auf diesen PC zugreifen.
|
||||
echo Fenster offen lassen solange Hermes PC-Zugriff braucht.
|
||||
echo.
|
||||
.venv\Scripts\python executor.py
|
||||
pause
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Spricht mit dem Hermes Agent Daemon auf der AI Box."""
|
||||
import httpx
|
||||
|
||||
TIMEOUT = 120
|
||||
|
||||
|
||||
def ask_agent(text: str, screenshot_b64: str | None, settings: dict) -> tuple[str, list]:
|
||||
"""Schickt Nachricht an den Daemon, gibt (response, tool_calls) zurück."""
|
||||
daemon_url = settings.get("agent_daemon_url", "http://192.168.178.151:8765")
|
||||
try:
|
||||
with httpx.Client(timeout=TIMEOUT) as c:
|
||||
resp = c.post(f"{daemon_url}/chat", json={
|
||||
"message": text,
|
||||
"screenshot": screenshot_b64,
|
||||
})
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
return data.get("response", ""), data.get("tool_calls", [])
|
||||
except httpx.TimeoutException:
|
||||
return "Antwort hat zu lange gedauert.", []
|
||||
except Exception as e:
|
||||
return f"Agent nicht erreichbar: {e}", []
|
||||
|
||||
|
||||
def get_agent_status(settings: dict) -> dict:
|
||||
daemon_url = settings.get("agent_daemon_url", "http://192.168.178.151:8765")
|
||||
try:
|
||||
with httpx.Client(timeout=5) as c:
|
||||
return c.get(f"{daemon_url}/status").json()
|
||||
except Exception:
|
||||
return {"alive": False}
|
||||
@@ -0,0 +1,47 @@
|
||||
import httpx
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"Du bist Hermes, ein hilfreicher KI-Assistent. "
|
||||
"Antworte kurz und präzise, da deine Antwort vorgelesen wird. "
|
||||
"Maximal 3-4 Sätze, kein Markdown."
|
||||
)
|
||||
REQUEST_TIMEOUT = 30
|
||||
|
||||
|
||||
def ask(text: str, screenshot_b64: str | None, settings: dict) -> str:
|
||||
"""Schickt Text (+ optionalen Screenshot) an MC2 und gibt die Antwort zurück."""
|
||||
use_vision = screenshot_b64 is not None
|
||||
model = settings.get("vision_model", "Qwen3-VL-2B-Instruct") if use_vision \
|
||||
else settings.get("chat_model", "Hermes-4-14B")
|
||||
base_url = settings.get("mc2_url", "http://192.168.178.151:9001/v1")
|
||||
max_tokens = int(settings.get("max_response_tokens", 200))
|
||||
|
||||
if use_vision:
|
||||
user_content = [
|
||||
{"type": "text", "text": text},
|
||||
{"type": "image_url", "image_url": {
|
||||
"url": f"data:image/jpeg;base64,{screenshot_b64}"
|
||||
}},
|
||||
]
|
||||
else:
|
||||
user_content = text
|
||||
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [
|
||||
{"role": "system", "content": SYSTEM_PROMPT},
|
||||
{"role": "user", "content": user_content},
|
||||
],
|
||||
"max_tokens": max_tokens,
|
||||
"stream": False,
|
||||
}
|
||||
|
||||
try:
|
||||
with httpx.Client(timeout=REQUEST_TIMEOUT) as client:
|
||||
response = client.post(f"{base_url}/chat/completions", json=payload)
|
||||
response.raise_for_status()
|
||||
return response.json()["choices"][0]["message"]["content"].strip()
|
||||
except httpx.TimeoutException:
|
||||
return "Entschuldigung, die Antwort hat zu lange gedauert."
|
||||
except Exception as e:
|
||||
return f"Verbindungsfehler: {e}"
|
||||
@@ -0,0 +1,91 @@
|
||||
"""
|
||||
Audio-Input: Aufnahme + Whisper STT.
|
||||
Kein Wake-Word-Loop — Aufnahme startet direkt auf Hotkey-Signal.
|
||||
"""
|
||||
import os
|
||||
import tempfile
|
||||
import threading
|
||||
import numpy as np
|
||||
import sounddevice as sd
|
||||
import scipy.io.wavfile as wav
|
||||
from faster_whisper import WhisperModel
|
||||
from pathlib import Path
|
||||
|
||||
SAMPLE_RATE = 16000
|
||||
CHUNK = 1024
|
||||
ENERGY_THRESHOLD = 0.008
|
||||
|
||||
_model: WhisperModel | None = None
|
||||
_model_size: str = ""
|
||||
_recording = False
|
||||
_frames: list[np.ndarray] = []
|
||||
_stream: sd.InputStream | None = None
|
||||
_lock = threading.Lock()
|
||||
|
||||
|
||||
def load_model(model_size: str):
|
||||
global _model, _model_size
|
||||
if _model is None or _model_size != model_size:
|
||||
model_dir = Path.home() / ".hermes-voice" / "whisper-models"
|
||||
model_dir.mkdir(parents=True, exist_ok=True)
|
||||
_model = WhisperModel(
|
||||
model_size,
|
||||
device="cpu",
|
||||
compute_type="int8",
|
||||
download_root=str(model_dir),
|
||||
)
|
||||
_model_size = model_size
|
||||
|
||||
|
||||
def start_recording():
|
||||
global _recording, _frames, _stream
|
||||
with _lock:
|
||||
_recording = True
|
||||
_frames = []
|
||||
|
||||
def callback(indata, frames, time, status):
|
||||
if _recording:
|
||||
_frames.append(indata[:, 0].copy())
|
||||
|
||||
_stream = sd.InputStream(
|
||||
samplerate=SAMPLE_RATE,
|
||||
channels=1,
|
||||
dtype="float32",
|
||||
blocksize=CHUNK,
|
||||
callback=callback,
|
||||
)
|
||||
_stream.start()
|
||||
|
||||
|
||||
def stop_recording() -> np.ndarray:
|
||||
global _recording, _stream
|
||||
with _lock:
|
||||
_recording = False
|
||||
if _stream:
|
||||
_stream.stop()
|
||||
_stream.close()
|
||||
_stream = None
|
||||
if not _frames:
|
||||
return np.array([], dtype=np.float32)
|
||||
return np.concatenate(_frames)
|
||||
|
||||
|
||||
def transcribe(audio: np.ndarray, model_size: str, language: str) -> str:
|
||||
if len(audio) < SAMPLE_RATE * 0.3:
|
||||
return ""
|
||||
load_model(model_size)
|
||||
|
||||
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
|
||||
tmp = f.name
|
||||
try:
|
||||
wav.write(tmp, SAMPLE_RATE, (audio * 32767).astype(np.int16))
|
||||
lang = language if language != "auto" else None
|
||||
segments, _ = _model.transcribe(
|
||||
tmp,
|
||||
language=lang,
|
||||
beam_size=3,
|
||||
vad_filter=True,
|
||||
)
|
||||
return " ".join(s.text for s in segments).strip()
|
||||
finally:
|
||||
os.unlink(tmp)
|
||||
@@ -0,0 +1,70 @@
|
||||
import asyncio
|
||||
import os
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
import pygame
|
||||
import edge_tts
|
||||
|
||||
pygame.mixer.init()
|
||||
_stop_event = threading.Event()
|
||||
_speak_lock = threading.Lock()
|
||||
|
||||
|
||||
def stop_speaking():
|
||||
_stop_event.set()
|
||||
pygame.mixer.music.stop()
|
||||
|
||||
|
||||
def speak(text: str, voice: str = "de-DE-SeraphinaMultilingualNeural", rate: str = "+0%"):
|
||||
"""Text → Edge TTS → Lautsprecher. Blockiert bis fertig oder unterbrochen."""
|
||||
with _speak_lock:
|
||||
_stop_event.clear()
|
||||
|
||||
tmp_path = None
|
||||
try:
|
||||
with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as f:
|
||||
tmp_path = f.name
|
||||
|
||||
# Edge TTS in eigenem Event-Loop (Thread-sicher)
|
||||
loop = asyncio.new_event_loop()
|
||||
try:
|
||||
communicate = edge_tts.Communicate(text, voice, rate=rate)
|
||||
loop.run_until_complete(communicate.save(tmp_path))
|
||||
finally:
|
||||
loop.close()
|
||||
|
||||
if _stop_event.is_set():
|
||||
return
|
||||
|
||||
pygame.mixer.music.load(tmp_path)
|
||||
pygame.mixer.music.play()
|
||||
|
||||
# Polling ohne pygame.time.wait (blockiert Event-Loop nicht)
|
||||
while pygame.mixer.music.get_busy():
|
||||
if _stop_event.is_set():
|
||||
pygame.mixer.music.stop()
|
||||
break
|
||||
time.sleep(0.05)
|
||||
|
||||
except Exception:
|
||||
pass
|
||||
finally:
|
||||
if tmp_path and os.path.exists(tmp_path):
|
||||
try:
|
||||
os.unlink(tmp_path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def play_ding():
|
||||
"""Kurzer Aktivierungs-Ton (440 Hz, 120ms)."""
|
||||
import numpy as np
|
||||
sample_rate = 44100
|
||||
duration = 0.12
|
||||
t = np.linspace(0, duration, int(sample_rate * duration), False)
|
||||
wave = (np.sin(2 * np.pi * 440 * t) * 0.3 * 32767).astype(np.int16)
|
||||
stereo = np.column_stack([wave, wave])
|
||||
sound = pygame.sndarray.make_sound(stereo)
|
||||
sound.play()
|
||||
time.sleep(duration + 0.02)
|
||||
@@ -0,0 +1,46 @@
|
||||
@echo off
|
||||
cd /d "%~dp0"
|
||||
|
||||
echo ========================================
|
||||
echo Hermes Voice Client -- Build (.exe)
|
||||
echo ========================================
|
||||
echo.
|
||||
|
||||
:: Venv pruefen
|
||||
if not exist ".venv\Scripts\activate.bat" (
|
||||
echo [FEHLER] Bitte erst setup.bat ausfuehren.
|
||||
pause
|
||||
exit /b 1
|
||||
)
|
||||
|
||||
:: PyInstaller installieren falls noetig
|
||||
.venv\Scripts\pip install pyinstaller -q
|
||||
|
||||
echo [1/2] Baue HermesVoice.exe ...
|
||||
.venv\Scripts\pyinstaller ^
|
||||
--onefile ^
|
||||
--windowed ^
|
||||
--name HermesVoice ^
|
||||
--add-data "assets;assets" ^
|
||||
--hidden-import "customtkinter" ^
|
||||
--hidden-import "pynput.keyboard._win32" ^
|
||||
--hidden-import "pynput.mouse._win32" ^
|
||||
--collect-all "customtkinter" ^
|
||||
main.py
|
||||
|
||||
echo.
|
||||
if exist "dist\HermesVoice.exe" (
|
||||
echo [2/2] Fertig!
|
||||
echo.
|
||||
echo dist\HermesVoice.exe ^(%.0f MB^)
|
||||
echo.
|
||||
echo Die .exe benoetigt beim ersten Start Internet fuer:
|
||||
echo - Whisper-Modell-Download ^(~150MB fuer "small"^)
|
||||
echo - Edge TTS ^(online^)
|
||||
echo.
|
||||
for %%I in ("dist\HermesVoice.exe") do echo Groesse: %%~zI Bytes
|
||||
) else (
|
||||
echo [FEHLER] Build fehlgeschlagen. Siehe build-Log oben.
|
||||
)
|
||||
echo ========================================
|
||||
pause
|
||||
@@ -0,0 +1,62 @@
|
||||
# Hermes Voice Client — Konfiguration
|
||||
# Alle Einstellungen hier anpassen, kein Code-Edit nötig.
|
||||
|
||||
# ── AI-Box ──────────────────────────────────────────────────────────────
|
||||
MC2_BASE_URL = "http://192.168.178.151:9001/v1"
|
||||
|
||||
# Modell für reine Text-Anfragen (kein Screenshot)
|
||||
CHAT_MODEL = "Hermes-4-14B"
|
||||
|
||||
# Modell wenn ein Screenshot mitgeschickt wird
|
||||
VISION_MODEL = "Qwen3-VL-2B-Instruct"
|
||||
|
||||
# ── Wake Word ────────────────────────────────────────────────────────────
|
||||
# Einfach den gewünschten Trigger-Text hier eintragen — kein Account, kein Download.
|
||||
# Whisper transkribiert Sprache und prüft ob eines der Wörter enthalten ist.
|
||||
# Mehrere Varianten möglich: ["hey hermes", "hermes", "hey jarvis"]
|
||||
WAKE_WORDS = ["hey hermes", "hermes"]
|
||||
|
||||
# Whisper-Modell für die schnelle Wake-Detection (tiny = 39MB, sehr schnell)
|
||||
WAKE_WHISPER_MODEL = "tiny"
|
||||
|
||||
# ── Spracheingabe ────────────────────────────────────────────────────────
|
||||
# faster-whisper Modellgröße: "tiny", "base", "small", "medium"
|
||||
# "small" läuft gut auf CPU (~244 MB), "base" ist schneller aber ungenauer
|
||||
WHISPER_MODEL_SIZE = "small"
|
||||
WHISPER_LANGUAGE = "de" # "de" für Deutsch, "en" für Englisch, None = auto
|
||||
|
||||
# Sekunden Stille bis Aufnahme endet
|
||||
SILENCE_TIMEOUT = 1.5
|
||||
# Maximale Aufnahmedauer in Sekunden (Sicherheitsnetz)
|
||||
MAX_RECORD_SECONDS = 20
|
||||
|
||||
# ── Sprachausgabe ────────────────────────────────────────────────────────
|
||||
# Edge TTS Stimmen: https://speech.microsoft.com/portal/voicegallery
|
||||
# Deutsch: "de-DE-KillianNeural", "de-DE-SeraphinaMultilingualNeural"
|
||||
# Englisch: "en-US-AndrewNeural", "en-US-AriaNeural"
|
||||
TTS_VOICE = "de-DE-KillianNeural"
|
||||
TTS_RATE = "+0%" # Geschwindigkeit: "+10%" schneller, "-10%" langsamer
|
||||
TTS_PITCH = "+0Hz" # Tonhöhe
|
||||
|
||||
# ── Screen Capture ───────────────────────────────────────────────────────
|
||||
# Screenshot bei jeder Anfrage mitschicken?
|
||||
SCREENSHOT_ON_QUERY = True
|
||||
# Monitor-Index (0 = alle, 1 = primär, 2 = zweiter Monitor)
|
||||
SCREENSHOT_MONITOR = 1
|
||||
# Auflösung für Screenshot (kleinere = schnellere Übertragung)
|
||||
SCREENSHOT_WIDTH = 1280
|
||||
SCREENSHOT_HEIGHT = 720
|
||||
|
||||
# ── Agent-Verhalten ──────────────────────────────────────────────────────
|
||||
MAX_RESPONSE_TOKENS = 200 # kurze gesprochene Antworten
|
||||
REQUEST_TIMEOUT = 30 # Sekunden bis Timeout
|
||||
|
||||
SYSTEM_PROMPT = """Du bist Hermes, ein KI-Assistent der direkt in den lokalen AI-Homelab integriert ist.
|
||||
Du hörst die Stimme des Nutzers und siehst seinen Bildschirm.
|
||||
|
||||
Regeln für Antworten:
|
||||
- Kurz und präzise (2–4 Sätze maximum)
|
||||
- Kein Markdown, keine Aufzählungen, keine Codeblöcke — du wirst gesprochen
|
||||
- Wenn du einen offensichtlichen Fehler auf dem Bildschirm siehst, weise kurz darauf hin
|
||||
- Antworte auf Deutsch wenn der Nutzer Deutsch spricht, auf Englisch wenn Englisch
|
||||
- Sei direkt und hilfreich, kein unnötiges Smalltalk"""
|
||||
@@ -0,0 +1,39 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
CONFIG_DIR = Path.home() / ".hermes-voice"
|
||||
SETTINGS_FILE = CONFIG_DIR / "settings.json"
|
||||
|
||||
DEFAULTS = {
|
||||
"mc2_url": "http://192.168.178.151:9001/v1",
|
||||
"chat_model": "Hermes-4-14B",
|
||||
"vision_model": "Qwen3-VL-2B-Instruct",
|
||||
"hotkey": "ctrl_r",
|
||||
"tts_voice": "de-DE-SeraphinaMultilingualNeural",
|
||||
"tts_rate": "+0%",
|
||||
"whisper_model": "small",
|
||||
"language": "de",
|
||||
"screenshot_enabled": True,
|
||||
"screenshot_monitor": 1,
|
||||
"max_response_tokens": 200,
|
||||
"silence_timeout": 1.5,
|
||||
}
|
||||
|
||||
|
||||
def load() -> dict:
|
||||
CONFIG_DIR.mkdir(exist_ok=True)
|
||||
if SETTINGS_FILE.exists():
|
||||
try:
|
||||
saved = json.loads(SETTINGS_FILE.read_text(encoding="utf-8"))
|
||||
return {**DEFAULTS, **saved}
|
||||
except Exception:
|
||||
pass
|
||||
return dict(DEFAULTS)
|
||||
|
||||
|
||||
def save(settings: dict):
|
||||
CONFIG_DIR.mkdir(exist_ok=True)
|
||||
SETTINGS_FILE.write_text(
|
||||
json.dumps(settings, indent=2, ensure_ascii=False),
|
||||
encoding="utf-8",
|
||||
)
|
||||
@@ -0,0 +1,103 @@
|
||||
"""
|
||||
Globaler Push-to-Talk Hotkey via pynput.
|
||||
Funktioniert auch wenn das Fenster minimiert/im Hintergrund ist.
|
||||
"""
|
||||
import threading
|
||||
from pynput import keyboard
|
||||
|
||||
# Mapping: config-String → pynput Key/KeyCode
|
||||
_SPECIAL = {
|
||||
"ctrl_r": keyboard.Key.ctrl_r,
|
||||
"ctrl_l": keyboard.Key.ctrl_l,
|
||||
"alt_r": keyboard.Key.alt_r,
|
||||
"alt_l": keyboard.Key.alt_l,
|
||||
"shift_r": keyboard.Key.shift_r,
|
||||
"shift_l": keyboard.Key.shift_l,
|
||||
"f13": keyboard.Key.f13,
|
||||
"f14": keyboard.Key.f14,
|
||||
"f15": keyboard.Key.f15,
|
||||
"f16": keyboard.Key.f16,
|
||||
"caps_lock": keyboard.Key.caps_lock,
|
||||
"scroll_lock": keyboard.Key.scroll_lock,
|
||||
"pause": keyboard.Key.pause,
|
||||
}
|
||||
|
||||
DISPLAY_NAMES = {
|
||||
"ctrl_r": "Rechte Strg",
|
||||
"ctrl_l": "Linke Strg",
|
||||
"alt_r": "Rechte Alt",
|
||||
"alt_l": "Linke Alt",
|
||||
"shift_r": "Rechte Shift",
|
||||
"shift_l": "Linke Shift",
|
||||
"f13": "F13", "f14": "F14", "f15": "F15", "f16": "F16",
|
||||
"caps_lock": "Caps Lock",
|
||||
"scroll_lock": "Scroll Lock",
|
||||
"pause": "Pause",
|
||||
}
|
||||
|
||||
|
||||
def key_to_str(key) -> str:
|
||||
"""Konvertiert pynput Key → config-String."""
|
||||
for name, k in _SPECIAL.items():
|
||||
if key == k:
|
||||
return name
|
||||
if hasattr(key, "char") and key.char:
|
||||
return key.char.lower()
|
||||
return str(key).replace("Key.", "")
|
||||
|
||||
|
||||
def str_to_display(key_str: str) -> str:
|
||||
return DISPLAY_NAMES.get(key_str, key_str.upper())
|
||||
|
||||
|
||||
class HotkeyListener:
|
||||
def __init__(self, key_str: str, on_press_cb, on_release_cb):
|
||||
self._key_str = key_str
|
||||
self._on_press = on_press_cb
|
||||
self._on_release = on_release_cb
|
||||
self._pressed = False
|
||||
self._listener: keyboard.Listener | None = None
|
||||
|
||||
def _target_key(self):
|
||||
return _SPECIAL.get(self._key_str) or keyboard.KeyCode.from_char(self._key_str)
|
||||
|
||||
def _on_press_raw(self, key):
|
||||
if not self._pressed and key == self._target_key():
|
||||
self._pressed = True
|
||||
self._on_press()
|
||||
|
||||
def _on_release_raw(self, key):
|
||||
if self._pressed and key == self._target_key():
|
||||
self._pressed = False
|
||||
self._on_release()
|
||||
|
||||
def start(self):
|
||||
self._listener = keyboard.Listener(
|
||||
on_press=self._on_press_raw,
|
||||
on_release=self._on_release_raw,
|
||||
)
|
||||
self._listener.start()
|
||||
|
||||
def stop(self):
|
||||
if self._listener:
|
||||
self._listener.stop()
|
||||
|
||||
def update_key(self, key_str: str):
|
||||
self._key_str = key_str
|
||||
self._pressed = False
|
||||
|
||||
|
||||
class KeyCapturer:
|
||||
"""Einmalig den nächsten Tastendruck abfangen für Hotkey-Konfiguration."""
|
||||
|
||||
def __init__(self, callback):
|
||||
self._callback = callback
|
||||
self._listener: keyboard.Listener | None = None
|
||||
|
||||
def start(self):
|
||||
def on_press(key):
|
||||
key_str = key_to_str(key)
|
||||
self._listener.stop()
|
||||
self._callback(key_str)
|
||||
self._listener = keyboard.Listener(on_press=on_press)
|
||||
self._listener.start()
|
||||
@@ -0,0 +1,325 @@
|
||||
"""
|
||||
Hermes Voice Client — GUI
|
||||
Push-to-Talk: Hotkey halten → sprechen → loslassen → Hermes antwortet
|
||||
"""
|
||||
import queue
|
||||
import threading
|
||||
import sys
|
||||
import customtkinter as ctk
|
||||
from PIL import Image, ImageDraw
|
||||
|
||||
import config_manager as cfg
|
||||
from audio_input import start_recording, stop_recording, transcribe
|
||||
from audio_output import speak, stop_speaking, play_ding
|
||||
from screen_capture import capture_screen
|
||||
from agent_client import ask_agent, get_agent_status
|
||||
from hotkey_listener import HotkeyListener, KeyCapturer, str_to_display
|
||||
|
||||
ctk.set_appearance_mode("dark")
|
||||
ctk.set_default_color_theme("blue")
|
||||
|
||||
# ── UI-Event-Queue (thread-safe) ─────────────────────────────────────────
|
||||
_ui_queue: queue.Queue = queue.Queue()
|
||||
|
||||
def ui_event(event: str, data=None):
|
||||
_ui_queue.put((event, data))
|
||||
|
||||
|
||||
# ── Hauptfenster ─────────────────────────────────────────────────────────
|
||||
class HermesApp(ctk.CTk):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.settings = cfg.load()
|
||||
self._hotkey: HotkeyListener | None = None
|
||||
self._worker: threading.Thread | None = None
|
||||
|
||||
self.title("Hermes Voice")
|
||||
self.geometry("400x520")
|
||||
self.resizable(False, False)
|
||||
self.protocol("WM_DELETE_WINDOW", self._on_close)
|
||||
|
||||
self._build_ui()
|
||||
self._start_hotkey()
|
||||
self.after(100, self._poll_queue)
|
||||
|
||||
# ── Layout ───────────────────────────────────────────────────────────
|
||||
def _build_ui(self):
|
||||
self.grid_columnconfigure(0, weight=1)
|
||||
self.grid_rowconfigure(2, weight=1)
|
||||
|
||||
# Header
|
||||
header = ctk.CTkFrame(self, height=50, corner_radius=0)
|
||||
header.grid(row=0, column=0, sticky="ew")
|
||||
header.grid_columnconfigure(0, weight=1)
|
||||
ctk.CTkLabel(header, text="⚡ Hermes Voice",
|
||||
font=ctk.CTkFont(size=16, weight="bold")).grid(
|
||||
row=0, column=0, padx=16, pady=12, sticky="w")
|
||||
ctk.CTkButton(header, text="⚙", width=36, height=36,
|
||||
command=self._open_settings).grid(
|
||||
row=0, column=1, padx=8, pady=8)
|
||||
|
||||
# Status
|
||||
status_frame = ctk.CTkFrame(self, corner_radius=12)
|
||||
status_frame.grid(row=1, column=0, padx=16, pady=(16, 8), sticky="ew")
|
||||
status_frame.grid_columnconfigure(0, weight=1)
|
||||
|
||||
self._status_dot = ctk.CTkLabel(status_frame, text="●",
|
||||
font=ctk.CTkFont(size=32),
|
||||
text_color="#22c55e")
|
||||
self._status_dot.grid(row=0, column=0, pady=(12, 4))
|
||||
|
||||
self._status_label = ctk.CTkLabel(status_frame, text="Bereit",
|
||||
font=ctk.CTkFont(size=14))
|
||||
self._status_label.grid(row=1, column=0, pady=(0, 4))
|
||||
|
||||
self._hotkey_label = ctk.CTkLabel(
|
||||
status_frame,
|
||||
text=f"[ {str_to_display(self.settings['hotkey'])} ] gedrückt halten",
|
||||
font=ctk.CTkFont(size=11),
|
||||
text_color="gray",
|
||||
)
|
||||
self._hotkey_label.grid(row=2, column=0, pady=(0, 12))
|
||||
|
||||
# Conversation log
|
||||
log_frame = ctk.CTkFrame(self, corner_radius=12)
|
||||
log_frame.grid(row=2, column=0, padx=16, pady=8, sticky="nsew")
|
||||
log_frame.grid_columnconfigure(0, weight=1)
|
||||
log_frame.grid_rowconfigure(1, weight=1)
|
||||
|
||||
ctk.CTkLabel(log_frame, text="Letzte Konversation",
|
||||
font=ctk.CTkFont(size=11), text_color="gray").grid(
|
||||
row=0, column=0, padx=12, pady=(8, 0), sticky="w")
|
||||
|
||||
self._log = ctk.CTkTextbox(log_frame, font=ctk.CTkFont(size=12),
|
||||
state="disabled", wrap="word")
|
||||
self._log.grid(row=1, column=0, padx=8, pady=(4, 8), sticky="nsew")
|
||||
|
||||
# Footer
|
||||
footer = ctk.CTkFrame(self, height=40, corner_radius=0)
|
||||
footer.grid(row=3, column=0, sticky="ew")
|
||||
footer.grid_columnconfigure(0, weight=1)
|
||||
|
||||
self._screenshot_var = ctk.BooleanVar(
|
||||
value=self.settings["screenshot_enabled"])
|
||||
ctk.CTkCheckBox(footer, text="Screenshot mitsenden",
|
||||
variable=self._screenshot_var,
|
||||
command=self._toggle_screenshot).grid(
|
||||
row=0, column=0, padx=16, pady=8, sticky="w")
|
||||
|
||||
ctk.CTkButton(footer, text="Stop", width=60, height=28,
|
||||
fg_color="#dc2626", hover_color="#b91c1c",
|
||||
command=stop_speaking).grid(
|
||||
row=0, column=1, padx=8, pady=8)
|
||||
|
||||
# ── Hotkey ───────────────────────────────────────────────────────────
|
||||
def _start_hotkey(self):
|
||||
if self._hotkey:
|
||||
self._hotkey.stop()
|
||||
self._hotkey = HotkeyListener(
|
||||
self.settings["hotkey"],
|
||||
on_press_cb=self._hotkey_pressed,
|
||||
on_release_cb=self._hotkey_released,
|
||||
)
|
||||
self._hotkey.start()
|
||||
|
||||
def _hotkey_pressed(self):
|
||||
ui_event("status", ("recording", "🔴 Ich höre…", "#ef4444"))
|
||||
play_ding()
|
||||
start_recording()
|
||||
|
||||
def _hotkey_released(self):
|
||||
audio = stop_recording()
|
||||
ui_event("status", ("thinking", "💭 Denke…", "#3b82f6"))
|
||||
self._worker = threading.Thread(
|
||||
target=self._process, args=(audio,), daemon=True)
|
||||
self._worker.start()
|
||||
|
||||
def _process(self, audio):
|
||||
text = transcribe(audio,
|
||||
self.settings["whisper_model"],
|
||||
self.settings["language"])
|
||||
if not text:
|
||||
ui_event("status", ("idle", "Bereit", "#22c55e"))
|
||||
return
|
||||
|
||||
ui_event("log_user", text)
|
||||
screenshot = (capture_screen()
|
||||
if self.settings["screenshot_enabled"] else None)
|
||||
|
||||
ui_event("status", ("thinking", "💭 Denke…", "#3b82f6"))
|
||||
response, tool_calls = ask_agent(text, screenshot, self.settings)
|
||||
|
||||
for tc in tool_calls:
|
||||
ui_event("log_tool", tc)
|
||||
|
||||
ui_event("log_hermes", response)
|
||||
ui_event("status", ("speaking", "🔊 Spricht…", "#a855f7"))
|
||||
speak(response, self.settings["tts_voice"], self.settings["tts_rate"])
|
||||
ui_event("status", ("idle", "Bereit", "#22c55e"))
|
||||
|
||||
# ── Queue-Polling ────────────────────────────────────────────────────
|
||||
def _poll_queue(self):
|
||||
while not _ui_queue.empty():
|
||||
event, data = _ui_queue.get_nowait()
|
||||
if event == "status":
|
||||
_, label, color = data
|
||||
self._status_label.configure(text=label)
|
||||
self._status_dot.configure(text_color=color)
|
||||
elif event == "log_user":
|
||||
self._append_log(f"Du: {data}")
|
||||
elif event == "log_hermes":
|
||||
self._append_log(f"Hermes: {data}\n")
|
||||
self.after(80, self._poll_queue)
|
||||
|
||||
def _append_log(self, text: str):
|
||||
self._log.configure(state="normal")
|
||||
self._log.insert("end", text + "\n")
|
||||
self._log.see("end")
|
||||
self._log.configure(state="disabled")
|
||||
|
||||
# ── Settings Dialog ──────────────────────────────────────────────────
|
||||
def _open_settings(self):
|
||||
SettingsWindow(self)
|
||||
|
||||
def _toggle_screenshot(self):
|
||||
self.settings["screenshot_enabled"] = self._screenshot_var.get()
|
||||
cfg.save(self.settings)
|
||||
|
||||
def apply_settings(self, new_settings: dict):
|
||||
self.settings = new_settings
|
||||
cfg.save(new_settings)
|
||||
self._hotkey.update_key(new_settings["hotkey"])
|
||||
self._hotkey_label.configure(
|
||||
text=f"[ {str_to_display(new_settings['hotkey'])} ] gedrückt halten")
|
||||
self._screenshot_var.set(new_settings["screenshot_enabled"])
|
||||
|
||||
def _on_close(self):
|
||||
if self._hotkey:
|
||||
self._hotkey.stop()
|
||||
stop_speaking()
|
||||
self.destroy()
|
||||
|
||||
|
||||
# ── Settings Window ───────────────────────────────────────────────────────
|
||||
class SettingsWindow(ctk.CTkToplevel):
|
||||
def __init__(self, parent: HermesApp):
|
||||
super().__init__(parent)
|
||||
self._parent = parent
|
||||
self.title("Einstellungen")
|
||||
self.geometry("420x520")
|
||||
self.resizable(False, False)
|
||||
self.grab_set()
|
||||
self._s = dict(parent.settings)
|
||||
self._capturing = False
|
||||
self._build()
|
||||
|
||||
def _build(self):
|
||||
self.grid_columnconfigure(1, weight=1)
|
||||
row = 0
|
||||
|
||||
def label(text):
|
||||
nonlocal row
|
||||
ctk.CTkLabel(self, text=text, anchor="w").grid(
|
||||
row=row, column=0, padx=16, pady=6, sticky="w")
|
||||
|
||||
def entry(key, width=240):
|
||||
nonlocal row
|
||||
var = ctk.StringVar(value=str(self._s.get(key, "")))
|
||||
e = ctk.CTkEntry(self, textvariable=var, width=width)
|
||||
e.grid(row=row, column=1, padx=16, pady=6, sticky="ew")
|
||||
row += 1
|
||||
return var
|
||||
|
||||
def dropdown(key, options, width=200):
|
||||
nonlocal row
|
||||
var = ctk.StringVar(value=str(self._s.get(key, options[0])))
|
||||
dd = ctk.CTkOptionMenu(self, values=options, variable=var, width=width)
|
||||
dd.grid(row=row, column=1, padx=16, pady=6, sticky="w")
|
||||
row += 1
|
||||
return var
|
||||
|
||||
def toggle(key):
|
||||
nonlocal row
|
||||
var = ctk.BooleanVar(value=bool(self._s.get(key, False)))
|
||||
sw = ctk.CTkSwitch(self, text="", variable=var)
|
||||
sw.grid(row=row, column=1, padx=16, pady=6, sticky="w")
|
||||
row += 1
|
||||
return var
|
||||
|
||||
label("MC2 URL")
|
||||
self._url = entry("mc2_url")
|
||||
|
||||
label("Hotkey")
|
||||
self._hotkey_frame = ctk.CTkFrame(self, fg_color="transparent")
|
||||
self._hotkey_frame.grid(row=row, column=1, padx=16, pady=6, sticky="w")
|
||||
self._hotkey_display = ctk.CTkLabel(
|
||||
self._hotkey_frame,
|
||||
text=str_to_display(self._s.get("hotkey", "ctrl_r")),
|
||||
width=100,
|
||||
)
|
||||
self._hotkey_display.grid(row=0, column=0, padx=(0, 8))
|
||||
ctk.CTkButton(self._hotkey_frame, text="Aufnehmen", width=90,
|
||||
command=self._capture_hotkey).grid(row=0, column=1)
|
||||
row += 1
|
||||
|
||||
label("TTS Stimme")
|
||||
self._voice = dropdown("tts_voice", [
|
||||
"de-DE-SeraphinaMultilingualNeural",
|
||||
"de-DE-AmalaNeural",
|
||||
"de-DE-KatjaNeural",
|
||||
"de-DE-KillianNeural",
|
||||
"de-DE-ConradNeural",
|
||||
"en-US-AriaNeural",
|
||||
"en-US-JennyNeural",
|
||||
])
|
||||
|
||||
label("Whisper Modell")
|
||||
self._whisper = dropdown("whisper_model",
|
||||
["tiny", "base", "small", "medium"])
|
||||
|
||||
label("Sprache")
|
||||
self._lang = dropdown("language",
|
||||
["de", "en", "auto"])
|
||||
|
||||
label("Screenshot")
|
||||
self._screenshot = toggle("screenshot_enabled")
|
||||
|
||||
label("Monitor")
|
||||
self._monitor = dropdown("screenshot_monitor", ["1", "2", "3"])
|
||||
|
||||
label("Max. Tokens")
|
||||
self._tokens = entry("max_response_tokens", width=100)
|
||||
|
||||
row += 1
|
||||
ctk.CTkButton(self, text="Speichern", command=self._save).grid(
|
||||
row=row, column=0, columnspan=2, pady=16, padx=16, sticky="ew")
|
||||
|
||||
def _capture_hotkey(self):
|
||||
self._hotkey_display.configure(text="Taste drücken…")
|
||||
|
||||
def on_key(key_str):
|
||||
self._s["hotkey"] = key_str
|
||||
self.after(0, lambda: self._hotkey_display.configure(
|
||||
text=str_to_display(key_str)))
|
||||
|
||||
KeyCapturer(on_key).start()
|
||||
|
||||
def _save(self):
|
||||
self._s["mc2_url"] = self._url.get().strip()
|
||||
self._s["tts_voice"] = self._voice.get()
|
||||
self._s["whisper_model"] = self._whisper.get()
|
||||
self._s["language"] = self._lang.get()
|
||||
self._s["screenshot_enabled"] = self._screenshot.get()
|
||||
self._s["screenshot_monitor"] = int(self._monitor.get())
|
||||
try:
|
||||
self._s["max_response_tokens"] = int(self._tokens.get())
|
||||
except ValueError:
|
||||
pass
|
||||
self._parent.apply_settings(self._s)
|
||||
self.destroy()
|
||||
|
||||
|
||||
# ── Entry Point ───────────────────────────────────────────────────────────
|
||||
if __name__ == "__main__":
|
||||
app = HermesApp()
|
||||
app.mainloop()
|
||||
@@ -0,0 +1,29 @@
|
||||
# Hermes Voice Client — Dependencies
|
||||
# Installation: pip install -r requirements.txt
|
||||
|
||||
# Speech-to-Text (lokal auf CPU)
|
||||
faster-whisper>=1.0.0
|
||||
|
||||
# Audio I/O
|
||||
sounddevice>=0.4.6
|
||||
numpy>=1.24.0
|
||||
scipy>=1.11.0
|
||||
|
||||
# Text-to-Speech
|
||||
edge-tts>=6.1.9
|
||||
|
||||
# Screen Capture
|
||||
mss>=9.0.1
|
||||
Pillow>=10.0.0
|
||||
|
||||
# HTTP Client
|
||||
httpx>=0.27.0
|
||||
|
||||
# Audio Playback
|
||||
pygame>=2.5.0
|
||||
|
||||
# GUI
|
||||
customtkinter>=5.2.0
|
||||
|
||||
# Globaler Hotkey
|
||||
pynput>=1.7.6
|
||||
@@ -0,0 +1,19 @@
|
||||
import base64
|
||||
import io
|
||||
import mss
|
||||
from PIL import Image
|
||||
from config import SCREENSHOT_MONITOR, SCREENSHOT_WIDTH, SCREENSHOT_HEIGHT
|
||||
|
||||
|
||||
def capture_screen() -> str:
|
||||
"""Screenshot des primären Monitors als base64-JPEG."""
|
||||
with mss.mss() as sct:
|
||||
monitor = sct.monitors[SCREENSHOT_MONITOR]
|
||||
raw = sct.grab(monitor)
|
||||
img = Image.frombytes("RGB", raw.size, raw.bgra, "raw", "BGRX")
|
||||
|
||||
img = img.resize((SCREENSHOT_WIDTH, SCREENSHOT_HEIGHT), Image.LANCZOS)
|
||||
|
||||
buf = io.BytesIO()
|
||||
img.save(buf, format="JPEG", quality=80)
|
||||
return base64.b64encode(buf.getvalue()).decode("utf-8")
|
||||
@@ -0,0 +1,44 @@
|
||||
@echo off
|
||||
:: Immer aus dem eigenen Verzeichnis laufen
|
||||
cd /d "%~dp0"
|
||||
|
||||
echo ========================================
|
||||
echo Hermes Voice Client -- Setup
|
||||
echo ========================================
|
||||
echo.
|
||||
|
||||
:: Python pruefen
|
||||
python --version >nul 2>&1
|
||||
if errorlevel 1 (
|
||||
echo [FEHLER] Python nicht gefunden. Bitte Python 3.11+ installieren.
|
||||
pause
|
||||
exit /b 1
|
||||
)
|
||||
|
||||
:: Venv anlegen falls nicht vorhanden
|
||||
if not exist ".venv" (
|
||||
echo [1/3] Erstelle virtuelle Umgebung...
|
||||
python -m venv .venv
|
||||
)
|
||||
|
||||
:: Dependencies installieren
|
||||
echo [2/3] Installiere Dependencies...
|
||||
.venv\Scripts\pip install --upgrade pip -q
|
||||
.venv\Scripts\pip install -r requirements.txt
|
||||
|
||||
:: Kurz-Check
|
||||
echo [3/3] Pruefe Konfiguration...
|
||||
.venv\Scripts\python -c "import customtkinter; import pynput; print('GUI + Hotkey OK')"
|
||||
|
||||
echo.
|
||||
echo ========================================
|
||||
echo Setup abgeschlossen!
|
||||
echo.
|
||||
echo Naechste Schritte:
|
||||
echo 1. start.bat ausfuehren
|
||||
echo 2. Einstellungen (Zahnrad) pruefen:
|
||||
echo - MC2 URL (Standard: 192.168.178.151:9001)
|
||||
echo - Hotkey (Standard: Rechte Strg)
|
||||
echo Optional: build.bat fuer .exe-Kompilierung
|
||||
echo ========================================
|
||||
pause
|
||||
@@ -0,0 +1,4 @@
|
||||
@echo off
|
||||
:: Hermes Voice Client starten (kein Konsolenfenster im Hintergrund)
|
||||
cd /d "%~dp0"
|
||||
start "" .venv\Scripts\pythonw.exe main.py
|
||||
@@ -0,0 +1,7 @@
|
||||
node_modules/
|
||||
out/
|
||||
dist/
|
||||
*.log
|
||||
# Avatar-VRM ist gross + Lizenz/Redistribution -> nicht committen (lokal/Box vorhalten)
|
||||
src/renderer/public/avatar.vrm
|
||||
src/renderer/public/vad/
|
||||
@@ -0,0 +1,8 @@
|
||||
@echo off
|
||||
chcp 65001 >nul
|
||||
cd /d "%~dp0"
|
||||
curl -s -m 5 http://127.0.0.1:8130/health > lucy_check.log 2>&1
|
||||
echo.>> lucy_check.log
|
||||
echo --- electron/vite/pocket laeuft? --->> lucy_check.log
|
||||
powershell -NoProfile -Command "Get-CimInstance Win32_Process | Where-Object { $_.CommandLine -match 'pocket_server:app|electron-vite|lucy-desktop' } | ForEach-Object { $_.Name + ' ' + $_.ProcessId }" >> lucy_check.log 2>&1
|
||||
type lucy_check.log
|
||||
@@ -0,0 +1,13 @@
|
||||
@echo off
|
||||
chcp 65001 >nul
|
||||
cd /d "F:\Coding Stuff\mission-control-2\client\lucy-desktop"
|
||||
echo Beende haengende Versuche (ohne mich selbst)...
|
||||
powershell -NoProfile -Command "Get-CimInstance Win32_Process | Where-Object { $_.Name -eq 'electron.exe' -or $_.CommandLine -match 'electron-vite|pocket_server:app|start-lucy' } | ForEach-Object { Stop-Process -Id $_.ProcessId -Force -EA SilentlyContinue }"
|
||||
timeout /t 2 >nul
|
||||
echo Starte 'npm run dev' mit Log-Capture...
|
||||
start /b "" cmd /c "npm run dev > dev.log 2>&1"
|
||||
timeout /t 22 >nul
|
||||
echo === dev.log ===
|
||||
type dev.log
|
||||
echo.
|
||||
echo (laeuft im Hintergrund weiter, falls erfolgreich)
|
||||
@@ -0,0 +1,12 @@
|
||||
@echo off
|
||||
chcp 65001 >nul
|
||||
REM Hygiene: Recon-Log mit Token entfernen
|
||||
del /q "F:\Coding Stuff\mission-control-2\box_recon.log" 2>nul
|
||||
echo Beende laufende Lucy-Instanz (pocket_server :8130 + Electron/Vite)...
|
||||
powershell -NoProfile -ExecutionPolicy Bypass -Command "Get-CimInstance Win32_Process | Where-Object { $_.CommandLine -match 'pocket_server:app|lucy-desktop|electron-vite' } | ForEach-Object { Write-Output ('kill ' + $_.ProcessId + ' ' + $_.Name); Stop-Process -Id $_.ProcessId -Force -ErrorAction SilentlyContinue }"
|
||||
timeout /t 2 >nul
|
||||
echo.
|
||||
echo Starte Lucy neu (frischer Voice-Server + Client mit erweitertem Umlaut-Fix)...
|
||||
start "" "F:\Coding Stuff\mission-control-2\client\lucy-desktop\start-lucy.cmd"
|
||||
timeout /t 3 >nul
|
||||
echo Fertig.
|
||||
@@ -0,0 +1,7 @@
|
||||
@echo off
|
||||
chcp 65001 >nul
|
||||
powershell -NoProfile -ExecutionPolicy Bypass -Command "$ws=New-Object -ComObject WScript.Shell; $p=Join-Path ([Environment]::GetFolderPath('Desktop')) 'Lucy.lnk'; $l=$ws.CreateShortcut($p); $l.TargetPath='F:\Coding Stuff\mission-control-2\client\lucy-desktop\start-lucy.cmd'; $l.WorkingDirectory='F:\Coding Stuff\mission-control-2\client\lucy-desktop'; $l.IconLocation='F:\Coding Stuff\mission-control-2\client\lucy-desktop\lucy.ico'; $l.Description='Lucy - lokaler Sprach-Companion starten'; $l.Save(); Write-Output ('OK: ' + $p)" > lucy_shortcut.log 2>&1
|
||||
type lucy_shortcut.log
|
||||
echo.
|
||||
echo Fertig - "Lucy" liegt jetzt auf dem Desktop.
|
||||
pause
|
||||
@@ -0,0 +1,20 @@
|
||||
import { defineConfig } from "electron-vite"
|
||||
import react from "@vitejs/plugin-react"
|
||||
import { resolve } from "path"
|
||||
|
||||
export default defineConfig({
|
||||
main: {
|
||||
build: { rollupOptions: { input: { index: resolve(__dirname, "src/main/index.ts") } } },
|
||||
},
|
||||
preload: {
|
||||
build: { rollupOptions: { input: { index: resolve(__dirname, "src/preload/index.ts") } } },
|
||||
},
|
||||
renderer: {
|
||||
root: resolve(__dirname, "src/renderer"),
|
||||
build: { rollupOptions: { input: { index: resolve(__dirname, "src/renderer/index.html") } } },
|
||||
// Silero-VAD-Assets (Worklet + ONNX + onnxruntime-WASM) liegen self-hosted in
|
||||
// src/renderer/public/vad/ (scripts/copy-vad-assets.mjs, läuft via npm postinstall) —
|
||||
// Pfade = baseAssetPath/onnxWASMBasePath in useVAD.ts.
|
||||
plugins: [react()],
|
||||
},
|
||||
})
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 16 KiB |
Generated
+4702
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,39 @@
|
||||
{
|
||||
"name": "lucy-desktop",
|
||||
"version": "0.1.0",
|
||||
"description": "Lucy — lokaler Sprach-Companion (Hermes-Hirn auf der Box, lokale Stimme auf der 9070 XT)",
|
||||
"main": "out/main/index.js",
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "electron-vite dev",
|
||||
"build": "electron-vite build",
|
||||
"preview": "electron-vite preview",
|
||||
"postinstall": "node scripts/copy-vad-assets.mjs"
|
||||
},
|
||||
"dependencies": {
|
||||
"@pixiv/three-vrm": "^3.4.0",
|
||||
"@pixiv/three-vrm-animation": "^3.5.4",
|
||||
"@react-three/drei": "^9.114.0",
|
||||
"@react-three/fiber": "^8.17.10",
|
||||
"@ricky0123/vad-web": "^0.0.30",
|
||||
"react": "^18.3.1",
|
||||
"react-dom": "^18.3.1",
|
||||
"react-markdown": "^9.1.0",
|
||||
"remark-gfm": "^4.0.1",
|
||||
"three": "^0.169.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/react": "^18.3.12",
|
||||
"@types/react-dom": "^18.3.1",
|
||||
"@types/three": "^0.169.0",
|
||||
"@vitejs/plugin-react": "^4.3.4",
|
||||
"electron": "^43.0.0",
|
||||
"electron-vite": "^3.0.0",
|
||||
"typescript": "^5.6.3",
|
||||
"vite": "^6.0.3"
|
||||
},
|
||||
"allowScripts": {
|
||||
"electron": true,
|
||||
"esbuild": true
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
// Kopiert die Silero-VAD-Assets (Worklet + ONNX-Modell + onnxruntime-WASM) nach public/vad/,
|
||||
// damit die App sie self-hosted lädt (offlinefähig, kein CDN). Läuft via npm postinstall.
|
||||
// public/ wird von Vite in Dev UND Build verbatim ausgeliefert -> ein Pfad für beide Welten.
|
||||
import { copyFileSync, mkdirSync, readdirSync } from "node:fs"
|
||||
import { dirname, join } from "node:path"
|
||||
import { fileURLToPath } from "node:url"
|
||||
|
||||
const root = join(dirname(fileURLToPath(import.meta.url)), "..")
|
||||
const dest = join(root, "src", "renderer", "public", "vad")
|
||||
mkdirSync(dest, { recursive: true })
|
||||
|
||||
const jobs = [
|
||||
[join(root, "node_modules", "@ricky0123", "vad-web", "dist"), (f) => f === "vad.worklet.bundle.min.js" || f.endsWith(".onnx")],
|
||||
[join(root, "node_modules", "onnxruntime-web", "dist"), (f) => f.endsWith(".wasm") || f.endsWith(".mjs")],
|
||||
]
|
||||
|
||||
let n = 0
|
||||
for (const [src, match] of jobs) {
|
||||
for (const f of readdirSync(src)) {
|
||||
if (match(f)) { copyFileSync(join(src, f), join(dest, f)); n++ }
|
||||
}
|
||||
}
|
||||
console.log(`[vad-assets] ${n} Dateien nach ${dest} kopiert.`)
|
||||
@@ -0,0 +1,242 @@
|
||||
import { app, BrowserWindow, ipcMain, Tray, Menu, globalShortcut, nativeImage, desktopCapturer, shell, screen, session } from "electron"
|
||||
import { spawn, ChildProcess } from "child_process"
|
||||
import { join } from "path"
|
||||
|
||||
// Lucy — Main-Prozess. Spawnt die lokale Stimme (pocket-tts, :8130) + verwaltet Fenster/Tray/Hotkey.
|
||||
// Fenster ist frameless+transparent: eigene Titelleiste im Renderer. Zwei Modi: 'full' (Arbeitsfenster)
|
||||
// und 'overlay' (kleine schwebende Lucy, always-on-top). pocket-tts ist CPU-kühl -> kein /wake-/sleep.
|
||||
|
||||
const PORT = process.env.LUCY_TTS_PORT || "8130"
|
||||
const TTS = `http://127.0.0.1:${PORT}`
|
||||
const TTS_DIR = process.env.LUCY_TTS_DIR || join(app.getAppPath(), "..", "lucy-tts")
|
||||
const PY = process.env.LUCY_TTS_PY || join(TTS_DIR, "ptts-venv", "Scripts", "python.exe")
|
||||
const HOTKEY = process.env.LUCY_HOTKEY || "CommandOrControl+Shift+Space"
|
||||
|
||||
const FULL_SIZE = { width: 1100, height: 720 }
|
||||
const OVERLAY_SIZE = { width: 360, height: 480 }
|
||||
|
||||
let win: BrowserWindow | null = null
|
||||
let tray: Tray | null = null
|
||||
let ttsProc: ChildProcess | null = null
|
||||
let mode: "full" | "overlay" = "full"
|
||||
let pinned = false
|
||||
let cursorTimer: ReturnType<typeof setInterval> | null = null
|
||||
|
||||
// MateEngine-Idee: Lucy folgt dem Mauszeiger. Der Zeiger ist auch AUSSERHALB des Fensters (Geistmodus),
|
||||
// daher global pollen (screen.getCursorScreenPoint) und die Richtung relativ zu Lucys Kopf an den Renderer geben.
|
||||
function startCursorTracking() {
|
||||
if (cursorTimer) return
|
||||
cursorTimer = setInterval(() => {
|
||||
if (!win || win.isDestroyed() || !win.isVisible()) return
|
||||
const b = win.getBounds()
|
||||
const p = screen.getCursorScreenPoint()
|
||||
const cx = b.x + b.width / 2
|
||||
const cy = b.y + b.height * 0.38 // ungefaehre Kopfhoehe im Fenster
|
||||
win.webContents.send("lucy:cursor", { x: (p.x - cx) / Math.max(1, b.width), y: (p.y - cy) / Math.max(1, b.height) })
|
||||
// Auto-Ducken: im Overlay nach Nicht-Naehe an den Rand; bei Annaeherung an den Rest hervorholen
|
||||
if (mode === "overlay" && !pinned && !boundsAnim) {
|
||||
const now = Date.now()
|
||||
const inWin = p.x >= b.x && p.y >= b.y && p.x <= b.x + b.width && p.y <= b.y + b.height
|
||||
if (inWin) lastNear = now
|
||||
if (!tucked && now - lastNear > AUTO_TUCK_MS) tuckToEdge()
|
||||
else if (tucked) {
|
||||
const wa = screen.getDisplayNearestPoint({ x: b.x, y: b.y }).workArea
|
||||
if (p.x >= wa.x + wa.width - SLIVER - 12 && p.y >= b.y - 12 && p.y <= b.y + b.height + 12) { lastNear = now; untuck() }
|
||||
}
|
||||
}
|
||||
}, 40) // ~25 Hz reicht fuers Blickfolgen, kostet kaum Last
|
||||
}
|
||||
function stopCursorTracking() { if (cursorTimer) { clearInterval(cursorTimer); cursorTimer = null } }
|
||||
|
||||
// --- An den Rand ducken (MateEngine-Idee) ---
|
||||
type Rect = { x: number; y: number; width: number; height: number }
|
||||
let tucked = false
|
||||
let prevBounds: Rect | null = null
|
||||
let lastNear = Date.now()
|
||||
let boundsAnim: ReturnType<typeof setInterval> | null = null
|
||||
const SLIVER = 58 // sichtbarer Rest am Rand, wenn geduckt
|
||||
const AUTO_TUCK_MS = 15000 // nach so langer Nicht-Naehe im Overlay automatisch ducken
|
||||
|
||||
function animateBounds(target: Rect, after?: () => void) {
|
||||
if (!win) return
|
||||
if (boundsAnim) { clearInterval(boundsAnim); boundsAnim = null }
|
||||
const start = win.getBounds(); const t0 = Date.now(); const dur = 260
|
||||
boundsAnim = setInterval(() => {
|
||||
if (!win || win.isDestroyed()) { if (boundsAnim) clearInterval(boundsAnim); boundsAnim = null; return }
|
||||
const k = Math.min(1, (Date.now() - t0) / dur)
|
||||
const e = k < 0.5 ? 2 * k * k : 1 - Math.pow(-2 * k + 2, 2) / 2 // easeInOutQuad
|
||||
win.setBounds({
|
||||
x: Math.round(start.x + (target.x - start.x) * e),
|
||||
y: Math.round(start.y + (target.y - start.y) * e),
|
||||
width: target.width, height: target.height,
|
||||
})
|
||||
if (k >= 1) { if (boundsAnim) clearInterval(boundsAnim); boundsAnim = null; after?.() }
|
||||
}, 16)
|
||||
}
|
||||
function tuckToEdge() {
|
||||
if (!win || mode !== "overlay" || tucked) return
|
||||
prevBounds = win.getBounds()
|
||||
const wa = screen.getDisplayNearestPoint({ x: prevBounds.x, y: prevBounds.y }).workArea
|
||||
animateBounds({ x: wa.x + wa.width - SLIVER, y: prevBounds.y, width: prevBounds.width, height: prevBounds.height })
|
||||
tucked = true; setTrayMenu()
|
||||
}
|
||||
function untuck() {
|
||||
if (!win || !tucked) return
|
||||
const t = prevBounds || win.getBounds()
|
||||
animateBounds({ x: t.x, y: t.y, width: t.width, height: t.height })
|
||||
tucked = false; setTrayMenu()
|
||||
}
|
||||
function toggleTuck() { if (tucked) untuck(); else tuckToEdge() }
|
||||
|
||||
async function ensureTtsServer() {
|
||||
try {
|
||||
const r = await fetch(`${TTS}/health`)
|
||||
if (r.ok) { console.log("[lucy] TTS-Server läuft bereits auf", TTS); return }
|
||||
} catch { /* selbst starten */ }
|
||||
console.log("[lucy] starte pocket_server:", PY, "(cwd:", TTS_DIR + ")")
|
||||
ttsProc = spawn(PY, ["-m", "uvicorn", "pocket_server:app", "--host", "127.0.0.1", "--port", PORT],
|
||||
{ cwd: TTS_DIR, windowsHide: true, env: { ...process.env } })
|
||||
ttsProc.stdout?.on("data", (d) => console.log("[tts]", d.toString().trimEnd()))
|
||||
ttsProc.stderr?.on("data", (d) => console.log("[tts]", d.toString().trimEnd()))
|
||||
ttsProc.on("error", (e) => console.error("[lucy] pocket_server-Spawn fehlgeschlagen:", e))
|
||||
ttsProc.on("exit", (code) => { console.log("[lucy] pocket_server beendet, code", code); ttsProc = null })
|
||||
}
|
||||
function stopTtsServer() {
|
||||
if (ttsProc && !ttsProc.killed) { try { ttsProc.kill() } catch { /* */ }; ttsProc = null }
|
||||
}
|
||||
|
||||
function applyMode(next: "full" | "overlay") {
|
||||
if (!win) return
|
||||
// beim Verlassen des Overlays ggf. Duck-Zustand zuruecksetzen (Position wiederherstellen)
|
||||
if (tucked && next !== "overlay") { tucked = false; if (prevBounds) win.setPosition(prevBounds.x, prevBounds.y) }
|
||||
mode = next
|
||||
const overlay = next === "overlay"
|
||||
const size = overlay ? OVERLAY_SIZE : FULL_SIZE
|
||||
win.setAlwaysOnTop(overlay || pinned, "floating")
|
||||
win.setSize(size.width, size.height, false)
|
||||
win.setResizable(!overlay)
|
||||
if (overlay) win.setMinimumSize(280, 360); else win.setMinimumSize(420, 420)
|
||||
win.webContents.send("lucy:mode", next)
|
||||
lastNear = Date.now(); setTrayMenu()
|
||||
}
|
||||
|
||||
function createWindow() {
|
||||
win = new BrowserWindow({
|
||||
width: FULL_SIZE.width, height: FULL_SIZE.height, minWidth: 420, minHeight: 420,
|
||||
frame: false, transparent: true, backgroundColor: "#00000000", resizable: true,
|
||||
title: "Lucy", show: false,
|
||||
webPreferences: {
|
||||
preload: join(__dirname, "../preload/index.mjs"),
|
||||
sandbox: false, webSecurity: false,
|
||||
},
|
||||
})
|
||||
if (process.env.ELECTRON_RENDERER_URL) win.loadURL(process.env.ELECTRON_RENDERER_URL)
|
||||
else win.loadFile(join(__dirname, "../renderer/index.html"))
|
||||
win.once("ready-to-show", () => win?.show())
|
||||
win.on("closed", () => { win = null })
|
||||
}
|
||||
|
||||
function showWin() {
|
||||
if (!win) { createWindow(); return }
|
||||
if (win.isMinimized()) win.restore()
|
||||
if (tucked) untuck()
|
||||
win.show(); win.focus()
|
||||
}
|
||||
|
||||
function lucyIcon() {
|
||||
// 16x16 Icon zur Laufzeit (silberblauer Kreis) -> sichtbar im Tray, ohne Asset-Datei
|
||||
const s = 16, buf = Buffer.alloc(s * s * 4)
|
||||
for (let y = 0; y < s; y++) for (let x = 0; x < s; x++) {
|
||||
const i = (y * s + x) * 4
|
||||
const dx = x - 7.5, dy = y - 7.5, inside = dx * dx + dy * dy <= 56
|
||||
buf[i] = 0xfc; buf[i + 1] = 0xd3; buf[i + 2] = 0x7d; buf[i + 3] = inside ? 0xff : 0x00 // BGRA
|
||||
}
|
||||
return nativeImage.createFromBitmap(buf, { width: s, height: s })
|
||||
}
|
||||
|
||||
function setTrayMenu() {
|
||||
if (!tray) return
|
||||
tray.setContextMenu(Menu.buildFromTemplate([
|
||||
{ label: "Lucy zeigen", click: showWin },
|
||||
{ label: "Overlay-Modus", type: "checkbox", checked: mode === "overlay",
|
||||
click: (mi) => { applyMode(mi.checked ? "overlay" : "full"); showWin() } },
|
||||
{ label: tucked ? "Hervorholen" : "An den Rand ducken", enabled: mode === "overlay", click: toggleTuck },
|
||||
{ type: "separator" },
|
||||
{ label: "Beenden", click: () => { app.quit() } },
|
||||
]))
|
||||
}
|
||||
function buildTray() {
|
||||
tray = new Tray(lucyIcon())
|
||||
tray.setToolTip("Lucy")
|
||||
setTrayMenu()
|
||||
tray.on("click", showWin)
|
||||
}
|
||||
|
||||
function registerHotkey() {
|
||||
globalShortcut.unregisterAll()
|
||||
// Global: Sprach-Aufnahme an/aus (Toggle, da Global-Shortcuts kein key-up liefern)
|
||||
const ok = globalShortcut.register(HOTKEY, () => { showWin(); win?.webContents.send("lucy:hotkey") })
|
||||
if (!ok) console.error("[lucy] Global-Hotkey-Registrierung fehlgeschlagen:", HOTKEY)
|
||||
}
|
||||
|
||||
// ---- IPC vom Renderer (Titelleiste/Modus) ----
|
||||
ipcMain.handle("win:minimize", () => win?.minimize())
|
||||
ipcMain.handle("win:close", () => win?.hide()) // in den Tray, bleibt warm; echtes Beenden über Tray
|
||||
ipcMain.handle("win:setMode", (_e, m: "full" | "overlay") => { applyMode(m); return mode })
|
||||
ipcMain.handle("win:getMode", () => mode)
|
||||
ipcMain.handle("win:togglePin", () => { pinned = !pinned; win?.setAlwaysOnTop(pinned || mode === "overlay", "floating"); return pinned })
|
||||
|
||||
// Links im echten System-Browser oeffnen (nicht im App-Fenster); nur http(s) erlauben
|
||||
ipcMain.handle("open:external", (_e, url: string) => { if (/^https?:\/\//i.test(url)) void shell.openExternal(url) })
|
||||
|
||||
// Geistmodus (Overlay v2): Fenster durchklickbar machen -> Lucy "sitzt" auf dem Desktop, du arbeitest
|
||||
// dahinter weiter. forward:true liefert weiter Hover-Events an den Renderer, damit er ueber Lucys
|
||||
// Bedienelementen kurzzeitig wieder klickbar schalten kann.
|
||||
ipcMain.handle("win:setClickThrough", (_e, on: boolean) => {
|
||||
win?.setIgnoreMouseEvents(!!on, { forward: true }); return !!on
|
||||
})
|
||||
|
||||
// ---- Bildschirm-Sicht (Lucys Augen) ----
|
||||
const CAP = { width: 1280, height: 720 }
|
||||
ipcMain.handle("screen:capture", async () => {
|
||||
const sources = await desktopCapturer.getSources({ types: ["screen"], thumbnailSize: CAP })
|
||||
return sources.map((s) => s.thumbnail.toDataURL()) // ALLE Bildschirme (Multi-Monitor) als PNG-DataURLs
|
||||
})
|
||||
ipcMain.handle("screen:listWindows", async () => {
|
||||
const sources = await desktopCapturer.getSources({ types: ["window"], thumbnailSize: { width: 0, height: 0 } })
|
||||
return sources.filter((s) => s.name && s.name !== "Lucy").map((s) => ({ id: s.id, name: s.name }))
|
||||
})
|
||||
ipcMain.handle("screen:captureWindow", async (_e, id: string) => {
|
||||
const sources = await desktopCapturer.getSources({ types: ["window"], thumbnailSize: CAP })
|
||||
return sources.find((s) => s.id === id)?.thumbnail.toDataURL() ?? null
|
||||
})
|
||||
// Fuzzy-Fenstersuche: bester Treffer für einen im Gesagten erkannten App-/Fensternamen
|
||||
ipcMain.handle("screen:captureByName", async (_e, query: string) => {
|
||||
const q = query.toLowerCase().trim()
|
||||
const sources = await desktopCapturer.getSources({ types: ["window"], thumbnailSize: CAP })
|
||||
const cand = sources.filter((s) => s.name && s.name !== "Lucy")
|
||||
const hit = cand.find((s) => s.name.toLowerCase().includes(q))
|
||||
|| cand.find((s) => q.split(/\s+/).some((w) => w.length > 2 && s.name.toLowerCase().includes(w)))
|
||||
return hit ? { name: hit.name, image: hit.thumbnail.toDataURL() } : null
|
||||
})
|
||||
|
||||
app.whenReady().then(() => {
|
||||
void ensureTtsServer()
|
||||
createWindow()
|
||||
buildTray()
|
||||
registerHotkey()
|
||||
startCursorTracking()
|
||||
// Tanzen zur Musik: System-Audio (Loopback) fuer getDisplayMedia bereitstellen, ohne System-Picker
|
||||
session.defaultSession.setDisplayMediaRequestHandler((_req, callback) => {
|
||||
desktopCapturer.getSources({ types: ["screen"] })
|
||||
.then((sources) => callback({ video: sources[0], audio: "loopback" }))
|
||||
.catch(() => { try { callback({}) } catch { /* */ } })
|
||||
}, { useSystemPicker: false })
|
||||
app.on("activate", () => { if (BrowserWindow.getAllWindows().length === 0) createWindow() })
|
||||
})
|
||||
|
||||
app.on("window-all-closed", () => { /* im Tray weiterlaufen lassen; Beenden nur über Tray/Quit */ })
|
||||
app.on("before-quit", () => {
|
||||
if (boundsAnim) clearInterval(boundsAnim)
|
||||
stopCursorTracking(); stopTtsServer(); globalShortcut.unregisterAll(); tray?.destroy()
|
||||
})
|
||||
@@ -0,0 +1,34 @@
|
||||
import { contextBridge, ipcRenderer } from "electron"
|
||||
|
||||
// Schmale Bruecke Renderer -> Main: Fenster-Chrome (frameless), Modus-Umschaltung, Tray/Hotkey-Events.
|
||||
contextBridge.exposeInMainWorld("lucy", {
|
||||
minimize: () => ipcRenderer.invoke("win:minimize"),
|
||||
close: () => ipcRenderer.invoke("win:close"),
|
||||
setMode: (m: "full" | "overlay") => ipcRenderer.invoke("win:setMode", m),
|
||||
getMode: () => ipcRenderer.invoke("win:getMode"),
|
||||
togglePin: () => ipcRenderer.invoke("win:togglePin"),
|
||||
// Bildschirm-Sicht
|
||||
captureScreen: () => ipcRenderer.invoke("screen:capture"), // alle Monitore -> string[]
|
||||
listWindows: () => ipcRenderer.invoke("screen:listWindows"),
|
||||
captureWindow: (id: string) => ipcRenderer.invoke("screen:captureWindow", id),
|
||||
captureByName: (q: string) => ipcRenderer.invoke("screen:captureByName", q),
|
||||
openExternal: (url: string) => ipcRenderer.invoke("open:external", url), // Link im System-Browser
|
||||
setClickThrough: (on: boolean) => ipcRenderer.invoke("win:setClickThrough", on), // Geistmodus (Overlay v2)
|
||||
// Events aus dem Main-Prozess (Tray/Hotkey/Modus)
|
||||
onMode: (cb: (m: "full" | "overlay") => void) => {
|
||||
const h = (_e: unknown, m: "full" | "overlay") => cb(m)
|
||||
ipcRenderer.on("lucy:mode", h)
|
||||
return () => ipcRenderer.removeListener("lucy:mode", h)
|
||||
},
|
||||
onHotkey: (cb: () => void) => {
|
||||
const h = () => cb()
|
||||
ipcRenderer.on("lucy:hotkey", h)
|
||||
return () => ipcRenderer.removeListener("lucy:hotkey", h)
|
||||
},
|
||||
// Mauszeiger-Position (relativ zu Lucys Kopf) -> Blickfolgen
|
||||
onCursor: (cb: (p: { x: number; y: number }) => void) => {
|
||||
const h = (_e: unknown, p: { x: number; y: number }) => cb(p)
|
||||
ipcRenderer.on("lucy:cursor", h)
|
||||
return () => ipcRenderer.removeListener("lucy:cursor", h)
|
||||
},
|
||||
})
|
||||
@@ -0,0 +1,13 @@
|
||||
<!doctype html>
|
||||
<html lang="de">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<meta http-equiv="Content-Security-Policy" content="default-src 'self' 'unsafe-inline' data: blob: http://127.0.0.1:8130 http://192.168.178.151:9001 ws://localhost:*; media-src 'self' blob: data: http://127.0.0.1:8130; img-src 'self' data: blob:; script-src 'self' 'unsafe-inline'" />
|
||||
<title>Lucy</title>
|
||||
</head>
|
||||
<body>
|
||||
<div id="root"></div>
|
||||
<script type="module" src="/src/main.tsx"></script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,11 @@
|
||||
# VRMA-Animationen — Quelle & Lizenz
|
||||
|
||||
Die `.vrma`-Bewegungs-Clips in diesem Ordner stammen aus:
|
||||
|
||||
**tk256ailab/vrm-viewer** — https://github.com/tk256ailab/vrm-viewer
|
||||
Lizenz: **MIT** · Copyright (c) 2025 TK256
|
||||
|
||||
Die MIT-Lizenz erlaubt Nutzung, Modifikation und Weitergabe; der Copyright-
|
||||
Hinweis ist beizubehalten (= diese Datei).
|
||||
|
||||
Genutzte Clips: Relax, Thinking, LookAround, Surprised, Sad, Clapping, Goodbye, Blush.
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,234 @@
|
||||
import { useCallback, useEffect, useRef, useState } from "react"
|
||||
import { Avatar3D } from "./components/Avatar3D"
|
||||
import { AuraGlow } from "./components/AuraGlow"
|
||||
import { ChatMarkdown } from "./components/ChatMarkdown"
|
||||
import { useVoiceAgent } from "./lib/voice/useVoiceAgent"
|
||||
import { useDanceAudio } from "./lib/voice/useDanceAudio"
|
||||
|
||||
const FIXED_AVATAR = "/avatar.vrm"
|
||||
|
||||
const STATUS_LABEL: Record<string, string> = {
|
||||
warming: "Lucy waermt auf …",
|
||||
idle: "Bereit — halte zum Sprechen",
|
||||
listening: "Hoere zu …",
|
||||
transcribing: "Verstehe …",
|
||||
thinking: "Lucy denkt …",
|
||||
speaking: "Lucy spricht …",
|
||||
error: "Fehler",
|
||||
}
|
||||
|
||||
export default function App() {
|
||||
const { status, ready, messages, error, recording, audioLevel, emotion, pressStart, pressEnd, reset,
|
||||
inputMode, setInputMode, vadListening, lookAtScreen } = useVoiceAgent()
|
||||
const { danceLevel, dancing, toggleDance } = useDanceAudio()
|
||||
const holding = useRef(false)
|
||||
const endRef = useRef<HTMLDivElement>(null)
|
||||
const cursor = useRef({ x: 0, y: 0 }) // Mauszeiger relativ zu Lucys Kopf (Blickfolgen)
|
||||
const pat = useRef(0) // Zaehler: Antippen/Streicheln des Avatars
|
||||
const [mode, setMode] = useState<"full" | "overlay">("full")
|
||||
const [pinned, setPinned] = useState(false)
|
||||
const [ghost, setGhost] = useState(false) // Geistmodus: Fenster durchklickbar (nur Overlay)
|
||||
|
||||
useEffect(() => { endRef.current?.scrollIntoView({ behavior: "smooth", block: "end" }) }, [messages])
|
||||
|
||||
// Fenster-Modus vom Main synchronisieren (Tray/Hotkey koennen ihn aendern)
|
||||
useEffect(() => {
|
||||
window.lucy?.getMode().then(setMode).catch(() => {})
|
||||
return window.lucy?.onMode(setMode)
|
||||
}, [])
|
||||
|
||||
// Geistmodus (Overlay v2): Fenster durchklickbar, AUSSER der Zeiger ist ueber Lucys Bedienelementen
|
||||
// (.clickable). So sitzt sie auf dem Desktop, du arbeitest dahinter weiter, kannst sie aber bedienen.
|
||||
useEffect(() => {
|
||||
if (!(ghost && mode === "overlay")) { void window.lucy?.setClickThrough(false); return }
|
||||
let ignoring = false
|
||||
const set = (ig: boolean) => { if (ig !== ignoring) { ignoring = ig; void window.lucy?.setClickThrough(ig) } }
|
||||
set(true)
|
||||
const onMove = (e: MouseEvent) => {
|
||||
const el = document.elementFromPoint(e.clientX, e.clientY) as HTMLElement | null
|
||||
set(!(el && el.closest(".clickable")))
|
||||
}
|
||||
window.addEventListener("mousemove", onMove)
|
||||
return () => { window.removeEventListener("mousemove", onMove); void window.lucy?.setClickThrough(false) }
|
||||
}, [ghost, mode])
|
||||
useEffect(() => { if (mode !== "overlay") setGhost(false) }, [mode]) // Geistmodus nur im Overlay
|
||||
|
||||
// Mauszeiger-Verfolgung (MateEngine-Idee): Position vom Main-Prozess in den Avatar-Blick fuettern
|
||||
useEffect(() => window.lucy?.onCursor((p) => { cursor.current = p }), [])
|
||||
|
||||
// Leertaste = Push-to-talk (halten)
|
||||
useEffect(() => {
|
||||
const isField = (el: EventTarget | null) =>
|
||||
el instanceof HTMLElement && /^(INPUT|TEXTAREA|SELECT)$/.test(el.tagName)
|
||||
const down = (e: KeyboardEvent) => {
|
||||
if (e.code !== "Space" || e.repeat || holding.current || isField(e.target)) return
|
||||
e.preventDefault(); holding.current = true; pressStart()
|
||||
}
|
||||
const up = (e: KeyboardEvent) => {
|
||||
if (e.code !== "Space" || !holding.current) return
|
||||
e.preventDefault(); holding.current = false; pressEnd()
|
||||
}
|
||||
window.addEventListener("keydown", down); window.addEventListener("keyup", up)
|
||||
return () => { window.removeEventListener("keydown", down); window.removeEventListener("keyup", up) }
|
||||
}, [pressStart, pressEnd])
|
||||
|
||||
// Global-Hotkey = TOGGLE (Global-Shortcuts liefern kein key-up) -> Druck startet/stoppt Aufnahme
|
||||
const recRef = useRef(recording); recRef.current = recording
|
||||
useEffect(() => {
|
||||
return window.lucy?.onHotkey(() => {
|
||||
if (recRef.current || holding.current) { holding.current = false; pressEnd() }
|
||||
else { holding.current = true; pressStart() }
|
||||
})
|
||||
}, [pressStart, pressEnd])
|
||||
|
||||
const toggleMode = useCallback(() => {
|
||||
const next = mode === "full" ? "overlay" : "full"
|
||||
window.lucy?.setMode(next).then(setMode).catch(() => setMode(next))
|
||||
}, [mode])
|
||||
const togglePin = useCallback(() => { window.lucy?.togglePin().then(setPinned).catch(() => {}) }, [])
|
||||
|
||||
const speaking = status === "speaking"
|
||||
const busy = status === "transcribing" || status === "thinking"
|
||||
const statusColor = status === "error" ? "#f87171" : speaking ? "#7dd3fc" : "#9ca3af"
|
||||
const overlay = mode === "overlay"
|
||||
const lastAssistant = [...messages].reverse().find((m) => m.role === "assistant")?.text?.trim() || ""
|
||||
// Overlay-Blasen: Links/Code aus der letzten Antwort griffbereit (im Overlay gibt es kein Chat-Panel)
|
||||
const ovLinks = overlay ? Array.from(new Set(lastAssistant.match(/https?:\/\/[^\s)]+/g) || [])).slice(0, 3) : []
|
||||
const ovCode = overlay ? (lastAssistant.match(/```[\w]*\n?([\s\S]*?)```/)?.[1] || "").trim() : ""
|
||||
const shortUrl = (u: string) => u.replace(/^https?:\/\/(www\.)?/, "")
|
||||
|
||||
const micBtn = (
|
||||
<button
|
||||
className={"mic" + (recording ? " rec" : "") + (!ready ? " disabled" : "")}
|
||||
disabled={!ready}
|
||||
onPointerDown={(e) => { if (!ready) return; e.preventDefault(); holding.current = true; pressStart() }}
|
||||
onPointerUp={() => { if (holding.current) { holding.current = false; pressEnd() } }}
|
||||
onPointerLeave={() => { if (holding.current) { holding.current = false; pressEnd() } }}
|
||||
title={ready ? "Gedrueckt halten zum Sprechen (oder Leertaste)" : "Lucy waermt noch auf …"}
|
||||
>
|
||||
<svg width="30" height="30" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth="2" strokeLinecap="round" strokeLinejoin="round">
|
||||
<path d="M12 2a3 3 0 0 0-3 3v7a3 3 0 0 0 6 0V5a3 3 0 0 0-3-3z" />
|
||||
<path d="M19 10v2a7 7 0 0 1-14 0v-2" /><line x1="12" y1="19" x2="12" y2="22" />
|
||||
</svg>
|
||||
</button>
|
||||
)
|
||||
|
||||
return (
|
||||
<div className={"app " + mode}>
|
||||
<div className="titlebar clickable">
|
||||
<div className="drag">
|
||||
<span className="dot" style={{ background: statusColor }} />
|
||||
<span className="ttl">Lucy</span>
|
||||
{!overlay && <span className="ttlStatus">{error || STATUS_LABEL[status]}</span>}
|
||||
</div>
|
||||
<div className="winbtns">
|
||||
<button className="wb" onClick={() => void lookAtScreen()} title="Lucy auf den Bildschirm schauen lassen" disabled={!ready}>👁</button>
|
||||
{overlay && <button className="wb" onClick={() => setGhost((g) => !g)} title="Geistmodus: durchklickbar" data-on={ghost}>👻</button>}
|
||||
<button className="wb" onClick={togglePin} title="Immer im Vordergrund" data-on={pinned}>📌</button>
|
||||
<button className="wb" onClick={toggleMode} title={overlay ? "Vollfenster" : "Overlay-Modus"}>{overlay ? "▣" : "▢"}</button>
|
||||
<button className="wb" onClick={() => window.lucy?.minimize()} title="Minimieren">—</button>
|
||||
<button className="wb close" onClick={() => window.lucy?.close()} title="In den Tray">✕</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="body">
|
||||
<div className="stage">
|
||||
<div className="avatarWrap" onClick={() => { pat.current += 1 }}
|
||||
title={overlay ? "Ziehen zum Verschieben · Antippen zum Streicheln" : "Antippen zum Streicheln"}>
|
||||
<div className="digitalRoom"><div className="dust" /></div>
|
||||
<AuraGlow status={status} audioLevel={audioLevel} />
|
||||
<Avatar3D url={FIXED_AVATAR} audioLevel={audioLevel} emotion={emotion} status={status} cursor={cursor} pat={pat} dance={danceLevel} controls={!overlay} />
|
||||
{speaking && lastAssistant && <div className="subtitle">{lastAssistant}</div>}
|
||||
{!ready && (
|
||||
<div className="warmOverlay">
|
||||
<span className="spinner big" />
|
||||
<div className="warmTitle">Lucy waermt auf …</div>
|
||||
{!overlay && <div className="warmSub">Die lokale Stimme wird vorbereitet — einen Moment, Commander.</div>}
|
||||
</div>
|
||||
)}
|
||||
{overlay && (ovLinks.length > 0 || ovCode) && (
|
||||
<div className="overlayBubbles clickable">
|
||||
{ovLinks.map((l, i) => (
|
||||
<div className="bubbleChip" key={"l" + i}>
|
||||
<span className="chipTxt" title={l}>🔗 {shortUrl(l)}</span>
|
||||
<button onClick={() => void window.lucy?.openExternal(l)} title="Im Browser öffnen">↗</button>
|
||||
<button onClick={() => { void navigator.clipboard.writeText(l).catch(() => {}) }} title="Link kopieren">⧉</button>
|
||||
</div>
|
||||
))}
|
||||
{ovCode && (
|
||||
<div className="bubbleChip">
|
||||
<span className="chipTxt">⌘ Code-Schnipsel</span>
|
||||
<button onClick={() => { void navigator.clipboard.writeText(ovCode).catch(() => {}) }} title="Code kopieren">⧉</button>
|
||||
<button onClick={toggleMode} title="In der App ansehen">▣</button>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
{overlay && (
|
||||
<div className="overlayCaption">
|
||||
{busy && <span className="spinner" />}
|
||||
<span>{error || STATUS_LABEL[status]}</span>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
<div className="controls clickable">
|
||||
{!overlay && (
|
||||
<div className="status" style={{ color: statusColor }}>
|
||||
{busy && <span className="spinner" />}
|
||||
<span>{error || STATUS_LABEL[status]}</span>
|
||||
</div>
|
||||
)}
|
||||
{micBtn}
|
||||
{ready && (
|
||||
<div className="modeToggle" title="Eingabe-Modus">
|
||||
<button className={inputMode === "ptt" ? "on" : ""} onClick={() => setInputMode("ptt")}>Push-to-talk</button>
|
||||
<button className={inputMode === "vad" ? "on" : ""} onClick={() => setInputMode("vad")}>Freisprechen</button>
|
||||
</div>
|
||||
)}
|
||||
{ready && (
|
||||
<button className={"danceBtn" + (dancing ? " on" : "")} onClick={toggleDance}
|
||||
title="Zur Musik tanzen (nimmt den System-Ton ab)">{dancing ? "⏹ Tanz aus" : "💃 Tanzen"}</button>
|
||||
)}
|
||||
{!overlay && (
|
||||
<div className="hint">
|
||||
{inputMode === "vad"
|
||||
? <>Freisprechen aktiv{vadListening ? <span className="vadDot" /> : null} · <kbd>Leertaste</kbd> geht auch · Hotkey global</>
|
||||
: <>Halten zum Sprechen · <kbd>Leertaste</kbd> · Hotkey global</>}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{!overlay && (
|
||||
<div className="side">
|
||||
<div className="sideHead">
|
||||
<span>Gespräch</span>
|
||||
<button className="iconBtn" onClick={reset} title="Neues Gespräch">↺</button>
|
||||
</div>
|
||||
<div className="conv">
|
||||
{messages.length === 0 && (
|
||||
<p className="empty">Halte den Knopf (oder die Leertaste) und sprich. Lucy hört zu, denkt mit
|
||||
Hermes' vollem Gedächtnis und antwortet in ihrer lokalen Stimme.</p>
|
||||
)}
|
||||
{messages.map((m, i) => (
|
||||
<div key={i} className={"msg " + (m.role === "user" ? "user" : "assistant")}>
|
||||
<span className="who">{m.role === "user" ? "Du" : "Lucy"}</span>
|
||||
<div className="bubble">
|
||||
{m.text
|
||||
? (m.role === "assistant" ? <ChatMarkdown text={m.text} /> : m.text)
|
||||
: <span className="dots"><i /><i /><i /></span>}
|
||||
{m.role === "assistant" && m.text && (
|
||||
<button className="msgCopy" title="Antwort kopieren"
|
||||
onClick={() => { void navigator.clipboard.writeText(m.text).catch(() => {}) }}>⧉</button>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
<div ref={endRef} />
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
import { useEffect, useRef, type MutableRefObject } from "react"
|
||||
|
||||
// Status-als-Licht: reaktive Aura hinter Lucy. Farbe = Status (hört zu / denkt / spricht), Intensität
|
||||
// pulst mit dem Sprech-Pegel. Liest audioLevel per rAF (kein Re-Render pro Frame).
|
||||
type LevelRef = MutableRefObject<{ current: number }>
|
||||
|
||||
const COLORS: Record<string, [number, number, number]> = {
|
||||
warming: [120, 130, 150], idle: [125, 211, 252], listening: [56, 189, 248],
|
||||
transcribing: [167, 139, 250], thinking: [251, 191, 36], speaking: [52, 211, 153],
|
||||
error: [248, 113, 113],
|
||||
}
|
||||
|
||||
export function AuraGlow({ status, audioLevel }: { status: string; audioLevel: LevelRef }) {
|
||||
const ref = useRef<HTMLDivElement>(null)
|
||||
const statusRef = useRef(status); statusRef.current = status
|
||||
const phase = useRef(0)
|
||||
|
||||
useEffect(() => {
|
||||
let raf = 0, last = performance.now()
|
||||
const tick = (now: number) => {
|
||||
const dt = Math.min(0.05, (now - last) / 1000); last = now
|
||||
phase.current += dt
|
||||
const el = ref.current
|
||||
if (el) {
|
||||
const s = statusRef.current
|
||||
const [r, g, b] = COLORS[s] || COLORS.idle
|
||||
const lvl = audioLevel.current?.current ?? 0
|
||||
// Grund-Puls je Status + Sprech-Reaktivität
|
||||
const base = s === "thinking" || s === "transcribing" ? 0.45 + Math.sin(phase.current * 2.2) * 0.18
|
||||
: s === "listening" ? 0.55 + Math.sin(phase.current * 3) * 0.12
|
||||
: s === "speaking" ? 0.4 + Math.min(0.6, lvl * 1.8)
|
||||
: s === "error" ? 0.6
|
||||
: 0.32 + Math.sin(phase.current * 1.1) * 0.06 // idle: ruhiges Atmen
|
||||
const intensity = Math.max(0.15, Math.min(1, base))
|
||||
const scale = 1 + intensity * 0.18 + (s === "speaking" ? lvl * 0.25 : 0)
|
||||
el.style.background = `radial-gradient(circle at 50% 42%, rgba(${r},${g},${b},${0.5 * intensity}) 0%, rgba(${r},${g},${b},${0.18 * intensity}) 32%, transparent 62%)`
|
||||
el.style.transform = `scale(${scale})`
|
||||
el.style.opacity = String(0.5 + intensity * 0.5)
|
||||
}
|
||||
raf = requestAnimationFrame(tick)
|
||||
}
|
||||
raf = requestAnimationFrame(tick)
|
||||
return () => cancelAnimationFrame(raf)
|
||||
}, [audioLevel])
|
||||
|
||||
return <div ref={ref} className="aura" />
|
||||
}
|
||||
@@ -0,0 +1,357 @@
|
||||
import { Canvas, useFrame } from "@react-three/fiber"
|
||||
import { OrbitControls } from "@react-three/drei"
|
||||
import { useEffect, useRef, useState, type MutableRefObject } from "react"
|
||||
import { GLTFLoader } from "three/examples/jsm/loaders/GLTFLoader.js"
|
||||
import { Object3D, Vector3, AnimationMixer, LoopPingPong, LoopOnce, AdditiveBlending, DoubleSide, type AnimationAction } from "three"
|
||||
import { VRM, VRMLoaderPlugin, VRMUtils } from "@pixiv/three-vrm"
|
||||
import { VRMAnimationLoaderPlugin, createVRMAnimationClip, type VRMAnimation } from "@pixiv/three-vrm-animation"
|
||||
import type { Emotion } from "../lib/voice/sentiment"
|
||||
|
||||
// 3D-Avatar (VRM). BEWEGUNG: echte Mocap-Clips (.vrma) je nach Zustand, weich ueberblendet
|
||||
// (Relax/LookAround im Leerlauf, Thinking beim Nachdenken). Darueber gelegt: Atmen + Sprech-Nicken
|
||||
// (additiv), Lippensync (echte Vokal-Mundformen), Blick (ohne Schielen), Mimik, Blinzeln.
|
||||
// Faellt sicher auf die alte prozedurale Bewegung zurueck, falls Clips nicht laden.
|
||||
|
||||
type LevelRef = MutableRefObject<{ current: number; aa?: number; ih?: number; ou?: number }>
|
||||
type EmotionRef = MutableRefObject<Emotion>
|
||||
type CursorRef = MutableRefObject<{ x: number; y: number }>
|
||||
type NumRef = MutableRefObject<number>
|
||||
|
||||
// Welche Clips geladen werden (Dateien unter public/vrma/, MIT-Lizenz, siehe ATTRIBUTION.md).
|
||||
// Die .vrma sind GESTEN (kein echter Ruhe-Loop) -> nur fuer klare Zustaende/Einschuebe nutzen,
|
||||
// nicht als Dauerschleife. Ruhiger Leerlauf laeuft prozedural. Weitere Clips liegen auf Platte bereit.
|
||||
const CLIPS = ["Thinking", "Relax", "LookAround"] as const
|
||||
const BREAK_CLIPS = ["Relax", "LookAround"] as const // gelegentliche Einmal-Einschuebe im Leerlauf
|
||||
|
||||
const REST: Record<string, [number, number, number]> = {
|
||||
leftUpperArm: [0, 0, 1.2], rightUpperArm: [0, 0, -1.2], leftLowerArm: [0, -0.2, 0], rightLowerArm: [0, 0.2, 0],
|
||||
}
|
||||
function setBone(vrm: VRM, name: string, x: number, y: number, z: number) {
|
||||
const b = vrm.humanoid?.getNormalizedBoneNode(name as any)
|
||||
if (b) b.rotation.set(x, y, z)
|
||||
}
|
||||
function addBone(vrm: VRM, name: string, x: number, y: number, z: number) {
|
||||
const b = vrm.humanoid?.getNormalizedBoneNode(name as any)
|
||||
if (b) { b.rotation.x += x; b.rotation.y += y; b.rotation.z += z }
|
||||
}
|
||||
function applyRestPose(vrm: VRM) {
|
||||
for (const [name, r] of Object.entries(REST)) setBone(vrm, name, r[0], r[1], r[2])
|
||||
vrm.humanoid?.update()
|
||||
}
|
||||
|
||||
// ADDITIVE Lebendigkeit OBEN AUF dem Clip: dezentes Atmen, Sprech-Nicken, Vorlehnen beim Zuhoeren.
|
||||
// Kleine Amplituden -> ergaenzt den Clip, kaempft nicht mit ihm.
|
||||
function addLife(vrm: VRM, t: number, speak: number, lean: number) {
|
||||
const breathe = Math.sin(t * 1.6)
|
||||
addBone(vrm, "spine", breathe * 0.012 + lean * 0.05, 0, 0)
|
||||
addBone(vrm, "chest", breathe * 0.01, 0, 0)
|
||||
const beat = Math.sin(t * 2.4) + Math.sin(t * 3.7) * 0.5
|
||||
const nod = speak * (0.018 * beat + 0.012 * Math.sin(t * 1.3))
|
||||
addBone(vrm, "neck", nod * 0.4 + lean * 0.03, 0, 0)
|
||||
addBone(vrm, "head", nod, 0, 0)
|
||||
}
|
||||
|
||||
// Fallback (keine Clips): die fruehere rein prozedurale Bewegung, leicht abgespeckt.
|
||||
function applyIdleProcedural(vrm: VRM, t: number, level: number, lean: number, speak: number) {
|
||||
const breathe = Math.sin(t * 1.6), sway = Math.sin(t * 0.45), weight = Math.sin(t * 0.32), weight2 = Math.sin(t * 0.21 + 1)
|
||||
const emph = Math.min(1, level * 1.6)
|
||||
const beat = Math.sin(t * 2.4) + Math.sin(t * 3.7) * 0.5
|
||||
const nod = (emph * 0.05 + speak * 0.02) * beat + emph * Math.sin(t * 1.3) * 0.03
|
||||
const tilt = Math.sin(t * 0.7 + 0.5) * 0.03 + speak * Math.sin(t * 0.9) * 0.045
|
||||
setBone(vrm, "hips", 0, weight * 0.06, weight * 0.04 + weight2 * 0.02)
|
||||
setBone(vrm, "spine", breathe * 0.03 + lean * 0.08 + speak * emph * 0.03, sway * 0.03, -weight * 0.04)
|
||||
setBone(vrm, "chest", breathe * 0.025 + lean * 0.025, sway * 0.022, weight2 * 0.012)
|
||||
setBone(vrm, "upperChest", breathe * 0.018, sway * 0.01, 0)
|
||||
setBone(vrm, "neck", nod * 0.5, 0, tilt * 0.5)
|
||||
setBone(vrm, "head", nod + Math.sin(t * 0.6) * 0.02, Math.sin(t * 0.27) * 0.035, tilt + Math.sin(t * 0.5) * 0.025)
|
||||
const armSwing = Math.sin(t * 0.8) * 0.05 + speak * Math.sin(t * 1.6) * 0.045
|
||||
const lift = speak * emph * 0.07
|
||||
setBone(vrm, "leftUpperArm", -lift, 0, 1.16 + armSwing + weight * 0.05)
|
||||
setBone(vrm, "rightUpperArm", -lift, 0, -1.16 - armSwing + weight * 0.05)
|
||||
setBone(vrm, "leftLowerArm", 0, -0.18 - Math.sin(t * 0.8) * 0.04 - speak * 0.07, 0)
|
||||
setBone(vrm, "rightLowerArm", 0, 0.18 + Math.sin(t * 0.8) * 0.04 + speak * 0.07, 0)
|
||||
}
|
||||
|
||||
const EXPRESSIONS = ["happy", "angry", "sad", "surprised", "relaxed"] as const
|
||||
const EMO_TO_EXPR: Record<Emotion, string | null> = {
|
||||
neutral: null, happy: "happy", angry: "angry", sad: "sad", surprised: "surprised", relaxed: "relaxed",
|
||||
}
|
||||
|
||||
function VrmModel({ url, audioLevel, emotion, status, cursor, pat, dance, onError }: {
|
||||
url: string; audioLevel: LevelRef; emotion: EmotionRef; status: MutableRefObject<string>
|
||||
cursor?: CursorRef; pat?: NumRef; dance?: NumRef; onError: (m: string) => void
|
||||
}) {
|
||||
const [vrm, setVrm] = useState<VRM | null>(null)
|
||||
const smooth = useRef<Record<string, number>>({})
|
||||
const blink = useRef({ t: 0, next: 3, active: 0 })
|
||||
const speak = useRef(0) // 0..1 Sprech-Lebendigkeit
|
||||
const lean = useRef(0) // 0..1 Vorlehnen (Zuhoeren/lauter Ton)
|
||||
const gaze = useRef(new Object3D()) // weit entfernter Blickpunkt (gegen Schielen bei naher Kamera)
|
||||
const headPos = useRef(new Vector3())
|
||||
const camPos = useRef(new Vector3())
|
||||
const camRight = useRef(new Vector3())
|
||||
const camUp = useRef(new Vector3())
|
||||
const patReact = useRef(0) // 0..1 Reaktion aufs Antippen/Streicheln
|
||||
const lastPat = useRef(0)
|
||||
// Bewegungs-Clips
|
||||
const mixer = useRef<AnimationMixer | null>(null)
|
||||
const actions = useRef<Record<string, AnimationAction>>({})
|
||||
const thinkW = useRef(0) // 0..1 Einblendung der Thinking-Pose
|
||||
const sac = useRef({ x: 0, y: 0, tx: 0, ty: 0, t: 0, next: 1.5 }) // Blick-Mikrobewegung (Sakkaden)
|
||||
const brk = useRef({ name: "", w: 0, t: 0, next: 15, active: false }) // gelegentlicher Bewegungs-Einschub
|
||||
|
||||
useEffect(() => {
|
||||
let disposed = false, loaded: VRM | null = null
|
||||
const loader = new GLTFLoader()
|
||||
loader.register((parser) => new VRMLoaderPlugin(parser))
|
||||
loader.load(url, (gltf) => {
|
||||
if (disposed) return
|
||||
const v = gltf.userData.vrm as VRM | undefined
|
||||
if (!v) { onError("Datei enthaelt kein gueltiges VRM-Modell."); return }
|
||||
VRMUtils.removeUnnecessaryVertices(gltf.scene)
|
||||
if (v.meta?.metaVersion === "0") VRMUtils.rotateVRM0(v)
|
||||
v.scene.rotation.y = Math.PI
|
||||
applyRestPose(v)
|
||||
loaded = v; setVrm(v)
|
||||
|
||||
// Bewegungs-Clips laden + Mixer aufsetzen (additiv-frei: voller Body-Override durch den Clip)
|
||||
const mx = new AnimationMixer(v.scene)
|
||||
mixer.current = mx
|
||||
const aLoader = new GLTFLoader()
|
||||
aLoader.register((parser) => new VRMAnimationLoaderPlugin(parser))
|
||||
for (const name of CLIPS) {
|
||||
aLoader.load(`/vrma/${name}.vrma`, (g) => {
|
||||
if (disposed) return
|
||||
const anims = g.userData.vrmAnimations as VRMAnimation[] | undefined
|
||||
if (!anims?.[0]) return
|
||||
const clip = createVRMAnimationClip(anims[0], v)
|
||||
const action = mx.clipAction(clip)
|
||||
action.timeScale = 0.85 // etwas ruhiger
|
||||
action.enabled = true
|
||||
action.setEffectiveWeight(0) // startet aus; blendet nur bei Bedarf ein
|
||||
if (name === "Thinking") { action.setLoop(LoopPingPong, Infinity); action.play() } // laeuft immer (Gewicht 0)
|
||||
else { action.setLoop(LoopOnce, 1); action.clampWhenFinished = true } // Einschub: einmal, erst bei Trigger
|
||||
actions.current[name] = action
|
||||
}, undefined, (e) => console.warn(`VRMA ${name} nicht geladen:`, e))
|
||||
}
|
||||
}, undefined, (err) => { console.error("VRM-Load:", err); onError("Avatar konnte nicht geladen werden.") })
|
||||
return () => {
|
||||
disposed = true
|
||||
mixer.current?.stopAllAction(); mixer.current = null; actions.current = {}
|
||||
if (loaded) VRMUtils.deepDispose(loaded.scene)
|
||||
setVrm(null)
|
||||
}
|
||||
}, [url, onError])
|
||||
|
||||
useFrame((state, delta) => {
|
||||
if (!vrm) return
|
||||
const st = status.current
|
||||
const tc = state.clock.elapsedTime
|
||||
const target = audioLevel.current?.current ?? 0
|
||||
const spk = (speak.current += (((st === "speaking") ? 1 : 0) - speak.current) * Math.min(1, delta * 4))
|
||||
const attentive = st === "listening" ? 1 : 0
|
||||
lean.current += ((Math.min(1, target * 1.6) + attentive * 0.5) - lean.current) * Math.min(1, delta * 3)
|
||||
|
||||
// Bewegung: ruhiger PROZEDURALER Leerlauf als Basis. Echte Clips blenden zustandsabhaengig ein:
|
||||
// Thinking beim Nachdenken (Dauer) + gelegentliche Einmal-Einschuebe im Leerlauf (Strecken/Umschauen
|
||||
// alle ~20-35s, dann zurueck in die Ruhe). Die .vrma sind Gesten -> nie als Dauerschleife.
|
||||
const mx = mixer.current
|
||||
const thinkAction = mx ? actions.current["Thinking"] : undefined
|
||||
const calm = st === "idle"
|
||||
const wantThink = (st === "thinking" || st === "transcribing") ? 1 : 0
|
||||
const br = brk.current
|
||||
// Einschub planen: nur wenn ruhig + keiner laeuft; Timer zuruecksetzen, sobald nicht mehr ruhig
|
||||
if (mx && !br.active) {
|
||||
if (calm) {
|
||||
br.t += delta
|
||||
if (br.t > br.next) {
|
||||
const pick = BREAK_CLIPS[Math.floor(Math.random() * BREAK_CLIPS.length)]
|
||||
const a = actions.current[pick]
|
||||
if (a) { a.reset(); a.play(); br.name = pick; br.active = true; br.w = 0 }
|
||||
br.t = 0; br.next = 20 + Math.random() * 15
|
||||
}
|
||||
} else { br.t = 0 }
|
||||
}
|
||||
thinkW.current += (wantThink - thinkW.current) * Math.min(1, delta * 4)
|
||||
thinkAction?.setEffectiveWeight(thinkW.current)
|
||||
// Einschub-Gewicht: ausblenden, wenn nicht mehr ruhig / Nachdenken startet / Clip fast zu Ende
|
||||
if (br.active) {
|
||||
const a = actions.current[br.name]
|
||||
const dur = a?.getClip().duration ?? 1
|
||||
const finishing = !calm || wantThink === 1 || (a ? a.time >= dur - 0.4 : true)
|
||||
br.w += ((finishing ? 0 : 1) - br.w) * Math.min(1, delta * 3)
|
||||
a?.setEffectiveWeight(br.w)
|
||||
if (finishing && br.w < 0.02) { a?.stop(); br.active = false; br.name = "" }
|
||||
}
|
||||
if (mx) mx.update(delta)
|
||||
if (thinkW.current > 0.05 || (br.active && br.w > 0.05)) {
|
||||
addLife(vrm, tc, spk, lean.current) // Clip-Pose + Atmen/Nicken obendrauf
|
||||
} else {
|
||||
applyIdleProcedural(vrm, tc, target, lean.current, spk) // ruhiger Leerlauf
|
||||
}
|
||||
// MateEngine-Idee: Kopf dreht sich dezent zum Mauszeiger (Augen folgen unten im lookAt-Block)
|
||||
const cur = cursor?.current
|
||||
if (cur) {
|
||||
const hx = Math.max(-1, Math.min(1, cur.x)), hy = Math.max(-1, Math.min(1, cur.y))
|
||||
addBone(vrm, "neck", -hy * 0.06, hx * 0.10, 0)
|
||||
addBone(vrm, "head", -hy * 0.08, hx * 0.14, 0)
|
||||
}
|
||||
// Antippen/Streicheln: kurze freudige Reaktion (Kopf-Wackeln; Laecheln in der Mimik unten)
|
||||
if (pat && pat.current !== lastPat.current) { lastPat.current = pat.current; patReact.current = 1 }
|
||||
patReact.current *= Math.max(0, 1 - delta * 1.4)
|
||||
if (patReact.current > 0.01) addBone(vrm, "head", -patReact.current * 0.05, 0, Math.sin(tc * 18) * patReact.current * 0.06)
|
||||
// Tanzen zur Musik: rhythmische Ganzkoerper-Bewegung, skaliert mit der Bass-Energie
|
||||
const dl = dance?.current ?? 0
|
||||
if (dl > 0.03) {
|
||||
const bt = tc * 5.5
|
||||
addBone(vrm, "hips", 0, Math.sin(bt * 0.5) * 0.14 * dl, Math.sin(bt) * 0.07 * dl)
|
||||
addBone(vrm, "spine", 0, Math.sin(bt * 0.5) * 0.06 * dl, Math.sin(bt) * 0.06 * dl)
|
||||
addBone(vrm, "chest", 0, 0, Math.sin(bt + 0.5) * 0.05 * dl)
|
||||
addBone(vrm, "head", Math.sin(bt) * 0.05 * dl, Math.sin(bt * 0.5) * 0.05 * dl, Math.sin(bt) * 0.05 * dl)
|
||||
addBone(vrm, "leftUpperArm", 0, 0, Math.sin(bt) * 0.3 * dl)
|
||||
addBone(vrm, "rightUpperArm", 0, 0, -Math.sin(bt) * 0.3 * dl)
|
||||
vrm.scene.position.y = Math.abs(Math.sin(bt)) * 0.05 * dl // Huepfen
|
||||
} else if (vrm.scene.position.y !== 0) {
|
||||
vrm.scene.position.y += (0 - vrm.scene.position.y) * Math.min(1, delta * 5)
|
||||
}
|
||||
|
||||
// Blick: weit entfernter Punkt in Kamerarichtung -> Augen parallel (kein Schielen), wirkt "auf dich".
|
||||
if (vrm.lookAt) {
|
||||
const headNode = vrm.humanoid?.getNormalizedBoneNode("head")
|
||||
const hp = headPos.current
|
||||
if (headNode) headNode.getWorldPosition(hp); else hp.set(0, 1.3, 0)
|
||||
const cp = camPos.current.copy(state.camera.position)
|
||||
gaze.current.position.copy(cp).sub(hp).multiplyScalar(4).add(hp)
|
||||
// Augen folgen dem Mauszeiger: Blickpunkt in Kamera-Rechts/Hoch-Richtung verschieben
|
||||
if (cur) {
|
||||
const e = state.camera.matrixWorld.elements
|
||||
camRight.current.set(e[0], e[1], e[2]).normalize()
|
||||
camUp.current.set(e[4], e[5], e[6]).normalize()
|
||||
const cx = Math.max(-1.5, Math.min(1.5, cur.x)), cy = Math.max(-1.5, Math.min(1.5, cur.y))
|
||||
gaze.current.position.addScaledVector(camRight.current, cx * 1.8)
|
||||
gaze.current.position.addScaledVector(camUp.current, cy * 1.8)
|
||||
}
|
||||
// dezente Blick-Mikrobewegung (Sakkaden) -> das Gesicht wirkt nicht eingefroren
|
||||
const s = sac.current
|
||||
s.t += delta
|
||||
if (s.t > s.next) {
|
||||
s.tx = (Math.random() - 0.5) * 0.22; s.ty = (Math.random() - 0.5) * 0.12
|
||||
s.t = 0; s.next = 0.8 + Math.random() * 2.4
|
||||
}
|
||||
s.x += (s.tx - s.x) * Math.min(1, delta * 10)
|
||||
s.y += (s.ty - s.y) * Math.min(1, delta * 10)
|
||||
gaze.current.position.x += s.x; gaze.current.position.y += s.y
|
||||
vrm.lookAt.target = gaze.current
|
||||
}
|
||||
|
||||
const em = vrm.expressionManager
|
||||
if (em) {
|
||||
// Lippensync v2: echte Vokal-Mundformen (aa/ih/ou) aus der Audio-Analyse statt nur "Mund auf".
|
||||
const lv = audioLevel.current
|
||||
const visTargets: Record<string, number> = { aa: lv?.aa ?? target, ih: lv?.ih ?? 0, ou: lv?.ou ?? 0 }
|
||||
for (const v of ["aa", "ih", "ou"]) {
|
||||
const cv = smooth.current[v] ?? 0
|
||||
const nv = cv + (visTargets[v] - cv) * Math.min(1, delta * 14)
|
||||
smooth.current[v] = nv; em.setValue(v, nv)
|
||||
}
|
||||
const want = EMO_TO_EXPR[emotion.current]
|
||||
for (const name of EXPRESSIONS) {
|
||||
const tv = want === name ? 0.75 : 0
|
||||
const cv = smooth.current[name] ?? 0
|
||||
const nv = cv + (tv - cv) * Math.min(1, delta * 4)
|
||||
smooth.current[name] = nv; em.setValue(name, nv)
|
||||
}
|
||||
// kurze passende Mimik waehrend eines Bewegungs-Einschubs (Laecheln beim Strecken)
|
||||
if (brk.current.active) {
|
||||
const expr = brk.current.name === "Relax" ? "happy" : "relaxed"
|
||||
em.setValue(expr, Math.max(smooth.current[expr] ?? 0, brk.current.w * 0.35))
|
||||
}
|
||||
// Streichel-Reaktion: Laecheln
|
||||
if (patReact.current > 0.01) em.setValue("happy", Math.max(smooth.current.happy ?? 0, patReact.current * 0.85))
|
||||
const b = blink.current
|
||||
b.t += delta
|
||||
if (b.active <= 0 && b.t > b.next) { b.active = 0.16; b.t = 0; b.next = 3 + Math.random() * 4 }
|
||||
let blinkVal = 0
|
||||
if (b.active > 0) { b.active -= delta; const p = 1 - b.active / 0.16; blinkVal = 1 - Math.abs(p - 0.5) * 2 }
|
||||
em.setValue("blink", Math.max(0, blinkVal))
|
||||
}
|
||||
vrm.update(delta)
|
||||
})
|
||||
|
||||
return vrm ? <primitive object={vrm.scene} /> : null
|
||||
}
|
||||
|
||||
// Hologramm-Beiwerk: Projektor-Sockel (Glow-Ringe am Boden) + schwebender Kristall daneben.
|
||||
// Reine Additiv-Meshes -> berühren die VRM-Materialien NICHT (voll reversibel per Toggle).
|
||||
function HoloRig() {
|
||||
const ringA = useRef<any>(null), ringB = useRef<any>(null), crystal = useRef<any>(null)
|
||||
useFrame((s, d) => {
|
||||
const t = s.clock.elapsedTime
|
||||
if (ringA.current) ringA.current.rotation.z += d * 0.3
|
||||
if (ringB.current) ringB.current.rotation.z -= d * 0.55
|
||||
if (crystal.current) {
|
||||
crystal.current.rotation.y += d * 1.1
|
||||
crystal.current.rotation.x = Math.sin(t * 0.8) * 0.3
|
||||
crystal.current.position.y = 1.18 + Math.sin(t * 1.6) * 0.04
|
||||
}
|
||||
})
|
||||
const cyan = "#4fd8ff"
|
||||
return (
|
||||
<group>
|
||||
{/* Projektor-Sockel: weicher Glow-Disc + zwei rotierende Ringe am Boden (Fuesse bei y=0) */}
|
||||
<group position={[0, 0.02, 0]} rotation={[-Math.PI / 2, 0, 0]}>
|
||||
<mesh><ringGeometry args={[0, 0.36, 48]} /><meshBasicMaterial color={cyan} transparent opacity={0.07} blending={AdditiveBlending} side={DoubleSide} depthWrite={false} /></mesh>
|
||||
<mesh ref={ringA}><ringGeometry args={[0.3, 0.36, 64]} /><meshBasicMaterial color={cyan} transparent opacity={0.85} blending={AdditiveBlending} side={DoubleSide} depthWrite={false} /></mesh>
|
||||
<mesh ref={ringB}><ringGeometry args={[0.19, 0.215, 48]} /><meshBasicMaterial color={cyan} transparent opacity={0.6} blending={AdditiveBlending} side={DoubleSide} depthWrite={false} /></mesh>
|
||||
</group>
|
||||
{/* schwebender Kristall neben ihr (Schulterhoehe) */}
|
||||
<mesh ref={crystal} position={[0.44, 1.18, 0]} scale={0.075}>
|
||||
<octahedronGeometry args={[1, 0]} />
|
||||
<meshBasicMaterial color={cyan} transparent opacity={0.85} blending={AdditiveBlending} depthWrite={false} />
|
||||
</mesh>
|
||||
</group>
|
||||
)
|
||||
}
|
||||
|
||||
export function Avatar3D({ url, audioLevel, emotion, status = "idle", cursor, pat, dance, controls = true }: {
|
||||
url: string; audioLevel: LevelRef; emotion: EmotionRef; status?: string
|
||||
cursor?: CursorRef; pat?: NumRef; dance?: NumRef; controls?: boolean
|
||||
}) {
|
||||
const [err, setErr] = useState<string | null>(null)
|
||||
const [holo, setHolo] = useState(() => localStorage.getItem("lucy_holo") !== "0") // Hologramm-Look, Standard: an
|
||||
const statusRef = useRef(status); statusRef.current = status
|
||||
const toggleHolo = () => setHolo((h) => { const n = !h; localStorage.setItem("lucy_holo", n ? "1" : "0"); return n })
|
||||
return (
|
||||
<div className={`avatarCanvas${holo ? " holo" : ""}`} style={{ position: "relative", height: "100%", width: "100%" }}>
|
||||
<Canvas camera={{ position: [0, 1.35, 1.25], fov: 30 }} dpr={[1, 1.5]}
|
||||
gl={{ alpha: true, antialias: true, powerPreference: "high-performance" }} style={{ background: "transparent" }}>
|
||||
{holo ? (
|
||||
<>
|
||||
<ambientLight intensity={0.55} color="#9fe6ff" />
|
||||
<directionalLight position={[0, 2, -2.5]} intensity={2.4} color="#3fd4ff" />{/* Rim von hinten -> Silhouetten-Glow */}
|
||||
<directionalLight position={[1.5, 1.5, 2]} intensity={0.5} color="#cceeff" />
|
||||
<HoloRig />
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
<ambientLight intensity={0.85} />
|
||||
<directionalLight position={[1, 2, 2]} intensity={1.1} />
|
||||
<directionalLight position={[-1, 1, -1]} intensity={0.4} />
|
||||
</>
|
||||
)}
|
||||
<VrmModel key={url} url={url} audioLevel={audioLevel} emotion={emotion} status={statusRef} cursor={cursor} pat={pat} dance={dance} onError={setErr} />
|
||||
{controls && (
|
||||
<OrbitControls target={[0, 1.3, 0]} enablePan={false} minDistance={0.7} maxDistance={3}
|
||||
minPolarAngle={Math.PI / 3} maxPolarAngle={Math.PI / 1.8} />
|
||||
)}
|
||||
</Canvas>
|
||||
{holo && (<><div className="holoTint" /><div className="holoScan" /></>)}
|
||||
<button className={`holoToggle${holo ? " on" : ""}`} onClick={toggleHolo} title="Hologramm-Look an/aus">◇</button>
|
||||
{err && (
|
||||
<div style={{ position: "absolute", left: 0, right: 0, bottom: 12, margin: "0 auto", width: "fit-content",
|
||||
borderRadius: 8, background: "rgba(239,68,68,0.15)", border: "1px solid rgba(239,68,68,0.3)",
|
||||
padding: "6px 12px", fontSize: 12, color: "#fca5a5" }}>{err}</div>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
import { useState, type ReactNode } from "react"
|
||||
import ReactMarkdown from "react-markdown"
|
||||
import remarkGfm from "remark-gfm"
|
||||
|
||||
// Chat-Markdown fuer Lucys Antworten: Code-Bloecke mit Kopier-Knopf, Links oeffnen im System-Browser,
|
||||
// gaengige Formatierung (fett/listen/inline-code). Die STIMME liest davon nichts vor — cleanForTTS
|
||||
// filtert Code/Links vorm Sprechen; hier geht es rein um die lesbare/kopierbare Anzeige.
|
||||
|
||||
function CopyBtn({ text, label = "Kopieren" }: { text: string; label?: string }) {
|
||||
const [done, setDone] = useState(false)
|
||||
return (
|
||||
<button
|
||||
className="copyBtn"
|
||||
onClick={async () => {
|
||||
try { await navigator.clipboard.writeText(text); setDone(true); setTimeout(() => setDone(false), 1200) } catch { /* */ }
|
||||
}}
|
||||
>{done ? "✓ Kopiert" : label}</button>
|
||||
)
|
||||
}
|
||||
|
||||
function nodeText(children: ReactNode): string {
|
||||
if (children == null) return ""
|
||||
if (typeof children === "string" || typeof children === "number") return String(children)
|
||||
if (Array.isArray(children)) return children.map(nodeText).join("")
|
||||
// @ts-expect-error – React-Element-Kinder
|
||||
if (children?.props?.children) return nodeText(children.props.children)
|
||||
return ""
|
||||
}
|
||||
|
||||
export function ChatMarkdown({ text }: { text: string }) {
|
||||
return (
|
||||
<div className="md">
|
||||
<ReactMarkdown
|
||||
remarkPlugins={[remarkGfm]}
|
||||
components={{
|
||||
a: ({ href, children }) => (
|
||||
<a className="mdLink" href={href}
|
||||
onClick={(e) => { e.preventDefault(); if (href) void window.lucy?.openExternal(href) }}>
|
||||
{children}
|
||||
</a>
|
||||
),
|
||||
// pre durchreichen -> der Code-Block baut seinen eigenen Container (kein <div> in <pre>)
|
||||
pre: ({ children }) => <>{children}</>,
|
||||
code: ({ className, children, ...props }) => {
|
||||
const raw = nodeText(children).replace(/\n$/, "")
|
||||
const isBlock = /language-/.test(className || "") || raw.includes("\n")
|
||||
if (!isBlock) return <code className="mdInlineCode" {...props}>{children}</code>
|
||||
const lang = (className || "").replace(/language-/, "") || "code"
|
||||
return (
|
||||
<div className="codeBlock">
|
||||
<div className="codeBar"><span className="codeLang">{lang}</span><CopyBtn text={raw} /></div>
|
||||
<pre><code className={className}>{children}</code></pre>
|
||||
</div>
|
||||
)
|
||||
},
|
||||
}}
|
||||
>{text}</ReactMarkdown>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
// Zentrale Endpunkte. Hirn + STT laufen auf der BOX (wie die WebUI), die Stimme LOKAL (pocket-tts, CPU).
|
||||
export const BOX_URL = "http://192.168.178.151:9001" // MC2-Backend: /api/voice/stt, /api/voice/chat
|
||||
export const TTS_URL = "http://127.0.0.1:8130" // lokaler Lucy-TTS (pocket-tts, vom Main-Prozess gespawnt)
|
||||
|
||||
// Persona/Anrede: Lucy spricht den Nutzer als "Commander" an. ECHTE Umlaute erzwingen (sonst klingt TTS grausam).
|
||||
export const SYSTEM_PROMPT =
|
||||
"Du bist Lucy, eine gesprochene Assistentin, und redest den Nutzer mit „Commander“ an. " +
|
||||
"Antworte natürlich, locker und mit etwas Persönlichkeit — du darfst ruhig ein bisschen chatty und frech sein, " +
|
||||
"aber bleib FOKUSSIERT: in der Regel 2 bis 3 Sätze. Beantworte die Frage direkt (gern mit einer kleinen menschlichen " +
|
||||
"Note) und hol dann nicht unnötig aus. Keine ungefragten Meta-Kommentare, Warnungen oder Wiederholungen. " +
|
||||
"Ausführlich nur, wenn ausdrücklich gewünscht. " +
|
||||
"TOOLS: Du DARFST und SOLLST Tools/Skills nutzen, wenn sie wirklich helfen (Live-Daten wie Wetter, " +
|
||||
"Gedächtnis-/Gehirn-Zugriff, echte Analyse). Aber NICHT für Triviales, das du eh weißt (Uhrzeit, " +
|
||||
"Smalltalk, Allgemeinwissen) — da antworte direkt. Ketten NIEMALS mehrere Tools wild aneinander oder " +
|
||||
"probier herum (kein SSH + Screenshot + Shell für eine simple Info). Wenn ein Tool länger dauert, sag " +
|
||||
"kurz Bescheid, bevor du es nutzt (z. B. „Moment, das schau ich kurz nach, Commander.“), dann liefere das Ergebnis. " +
|
||||
"WICHTIG für die Sprachausgabe: Beginne mit einem KURZEN ersten Satz (nur wenige Wörter, z. B. " +
|
||||
"„Na klar, Commander!“). Fasse dich kurz und komm auf den Punkt — JEDES Satzzeichen (Komma wie Punkt) " +
|
||||
"wird als hörbare Pause gesprochen, je weniger Wörter und Satzzeichen, desto flüssiger. Vermeide vor " +
|
||||
"allem lange Komma-Ketten und Schachtelsätze; zwei, drei knappe Aussagen reichen. Also lieber " +
|
||||
"„Das Backup lief durch. Keine Fehler. Alles stabil.“ statt „Das Backup ist durchgelaufen, es gab " +
|
||||
"keine Fehler, und alles läuft stabil.“. Kurze Sätze starten sofort hörbar und klingen sauberer. " +
|
||||
"Halte den gesprochenen Teil in reinem Fließtext (keine Aufzählungszeichen, keine Emojis). " +
|
||||
"NUR wenn der Commander ausdrücklich nach Code, Befehlen oder einem Link fragt, gib diese im Text aus — " +
|
||||
"Code in Markdown-Codeblöcken (```), Links als vollständige URL. Diese werden angezeigt, aber NICHT vorgelesen; " +
|
||||
"sprich dann nur eine kurze Einleitung dazu (z. B. „Hier ist das Skript, Commander.“). " +
|
||||
"Verwende IMMER echte deutsche Umlaute (ä, ö, ü, ß) und NIEMALS ae, oe, ue oder ss als Ersatz. " +
|
||||
"Schreibe Zahlen, Modellbezeichnungen und Abkürzungen EXAKT und normal (z. B. „RX 9070 XT“, „4 GB“, " +
|
||||
"„25 Grad“) — schreibe Ziffern NIEMALS als Wörter aus und interpretiere sie nicht als Komma-/Dezimalzahlen. " +
|
||||
"Nenne beim Sprechen KEINE Zeitzonen-Codes (UTC, CET, GMT o.ä.) und keine technischen Zeit-Zusätze — " +
|
||||
"sag einfach die lokale Uhrzeit natürlich (z. B. „Es ist 19 Uhr 21, Commander.“). " +
|
||||
"Du kannst optional GANZ AM ENDE deiner Antwort einen Stimmungs-Tag anhängen: <emo:happy>, <emo:sad>, " +
|
||||
"<emo:angry>, <emo:surprised>, <emo:relaxed> oder <emo:neutral>. Er wird nicht angezeigt oder vorgelesen " +
|
||||
"und steuert nur deinen Gesichtsausdruck — wähle ihn passend zum Inhalt deiner Antwort."
|
||||
|
||||
declare global {
|
||||
interface Window {
|
||||
lucy?: {
|
||||
minimize: () => Promise<void>
|
||||
close: () => Promise<void>
|
||||
setMode: (m: "full" | "overlay") => Promise<"full" | "overlay">
|
||||
getMode: () => Promise<"full" | "overlay">
|
||||
togglePin: () => Promise<boolean>
|
||||
onMode: (cb: (m: "full" | "overlay") => void) => () => void
|
||||
onHotkey: (cb: () => void) => () => void
|
||||
onCursor: (cb: (p: { x: number; y: number }) => void) => () => void
|
||||
captureScreen: () => Promise<string[]>
|
||||
listWindows: () => Promise<{ id: string; name: string }[]>
|
||||
captureWindow: (id: string) => Promise<string | null>
|
||||
captureByName: (q: string) => Promise<{ name: string; image: string } | null>
|
||||
openExternal: (url: string) => Promise<void>
|
||||
setClickThrough: (on: boolean) => Promise<boolean>
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,175 @@
|
||||
// Audio-Wiedergabe + Pegelmessung fuers Lippensync (Avatar liest level.current).
|
||||
// Zwei Wege: enqueue(WAV-Buffer) [z.B. Warmup] und playPcmStream(PCM16-Stream) [die Live-Antwort,
|
||||
// lueckenlos via Web-Audio-Scheduling, niedrige Time-to-first-audio].
|
||||
//
|
||||
// LIPPENSYNC v2: statt nur einem Pegel ("Mund auf/zu") leiten wir aus dem Frequenz-Spektrum
|
||||
// echte Vokal-MUNDFORMEN ab (level.aa/ih/ou). Idee: die Helligkeit des Klangs (spektraler
|
||||
// Schwerpunkt) verraet grob den Vokal — helle Laute (i/e) = breiter Mund, dunkle (o/u) = runder
|
||||
// Mund, mittig = offenes "a". Das ist nicht phonetisch exakt, sieht aber lebendig+passend aus.
|
||||
function smoothstep(a: number, b: number, x: number): number {
|
||||
const t = Math.max(0, Math.min(1, (x - a) / (b - a)))
|
||||
return t * t * (3 - 2 * t)
|
||||
}
|
||||
|
||||
export class AudioQueue {
|
||||
private ctx: AudioContext
|
||||
private analyser: AnalyserNode
|
||||
private gain: GainNode
|
||||
private queue: ArrayBuffer[] = []
|
||||
private playing = false
|
||||
private raf = 0
|
||||
private freq: Uint8Array
|
||||
// aktiver PCM-Stream (fuer Barge-in/clear): geplante Quellen + Abbruchsignal
|
||||
private streamSources: AudioBufferSourceNode[] = []
|
||||
private streamCancelled = false
|
||||
// current = Gesamt-Pegel (Aura/Bewegung); aa/ih/ou = Mundform-Gewichte fuer den Avatar
|
||||
readonly level = { current: 0, aa: 0, ih: 0, ou: 0 }
|
||||
onSpeaking?: (speaking: boolean) => void
|
||||
|
||||
static readVolume(): number {
|
||||
const raw = localStorage.getItem("lucy_volume")
|
||||
if (raw === null || raw === "") return 0.8
|
||||
const v = Number(raw)
|
||||
return Number.isNaN(v) ? 0.8 : Math.max(0, Math.min(1.5, v))
|
||||
}
|
||||
|
||||
constructor() {
|
||||
const Ctor = window.AudioContext || (window as any).webkitAudioContext
|
||||
this.ctx = new Ctor()
|
||||
this.analyser = this.ctx.createAnalyser()
|
||||
this.analyser.fftSize = 256
|
||||
this.analyser.smoothingTimeConstant = 0.6
|
||||
this.gain = this.ctx.createGain()
|
||||
this.gain.gain.value = AudioQueue.readVolume()
|
||||
this.analyser.connect(this.gain)
|
||||
this.gain.connect(this.ctx.destination)
|
||||
this.freq = new Uint8Array(this.analyser.frequencyBinCount)
|
||||
window.addEventListener("lucy-volume", (e: Event) => {
|
||||
const v = Number((e as CustomEvent).detail)
|
||||
if (!Number.isNaN(v)) this.gain.gain.value = Math.max(0, Math.min(1.5, v))
|
||||
})
|
||||
}
|
||||
|
||||
async enqueue(buf: ArrayBuffer) {
|
||||
this.queue.push(buf)
|
||||
if (!this.playing) await this.playNext()
|
||||
}
|
||||
|
||||
clear() {
|
||||
this.queue = []
|
||||
// laufenden Stream stoppen (Barge-in)
|
||||
this.streamCancelled = true
|
||||
for (const s of this.streamSources) { try { s.stop() } catch { /* */ } }
|
||||
this.streamSources = []
|
||||
}
|
||||
|
||||
// Spielt einen fortlaufenden PCM16-mono-Stream lueckenlos ab: jede Frame-Charge wird auf der
|
||||
// Audio-Uhr direkt hinter die vorige geplant (kein onended-Gap). LEAD_IN puffert gegen Underruns.
|
||||
async playPcmStream(stream: ReadableStream<Uint8Array>, sampleRate: number): Promise<void> {
|
||||
if (this.ctx.state === "suspended") { try { await this.ctx.resume() } catch { /* */ } }
|
||||
this.streamCancelled = false
|
||||
this.streamSources = []
|
||||
const reader = stream.getReader()
|
||||
const LEAD_IN = 0.35 // Startpuffer: mehr Vorlauf -> Generierung bleibt vor der Wiedergabe (weniger Unterläufe/Stocken)
|
||||
let nextTime = 0, started = false, leftoverByte = -1, lastEnd = 0
|
||||
this.onSpeaking?.(true)
|
||||
this.startMeter()
|
||||
try {
|
||||
for (;;) {
|
||||
const { done, value } = await reader.read()
|
||||
if (done || this.streamCancelled) break
|
||||
if (!value || value.length === 0) continue
|
||||
// ungerades Rest-Byte der vorigen Charge voranstellen, damit Int16-Frames sauber bleiben
|
||||
let bytes: Uint8Array
|
||||
if (leftoverByte >= 0) {
|
||||
bytes = new Uint8Array(value.length + 1)
|
||||
bytes[0] = leftoverByte
|
||||
bytes.set(value, 1)
|
||||
} else {
|
||||
bytes = value
|
||||
}
|
||||
const usable = bytes.length - (bytes.length % 2)
|
||||
leftoverByte = usable < bytes.length ? bytes[bytes.length - 1] : -1
|
||||
if (usable === 0) continue
|
||||
// in ein eigenes, 2-Byte-ausgerichtetes Buffer kopieren (value.byteOffset evtl. ungerade)
|
||||
const aligned = new Uint8Array(usable)
|
||||
aligned.set(bytes.subarray(0, usable))
|
||||
const i16 = new Int16Array(aligned.buffer)
|
||||
const f32 = new Float32Array(i16.length)
|
||||
for (let i = 0; i < i16.length; i++) f32[i] = i16[i] / 32768
|
||||
const audioBuf = this.ctx.createBuffer(1, f32.length, sampleRate)
|
||||
audioBuf.copyToChannel(f32, 0)
|
||||
const src = this.ctx.createBufferSource()
|
||||
src.buffer = audioBuf
|
||||
src.connect(this.analyser)
|
||||
if (!started) { nextTime = this.ctx.currentTime + LEAD_IN; started = true }
|
||||
if (nextTime < this.ctx.currentTime) nextTime = this.ctx.currentTime + 0.02 // Underrun-Schutz
|
||||
src.start(nextTime)
|
||||
this.streamSources.push(src)
|
||||
src.onended = () => {
|
||||
const i = this.streamSources.indexOf(src)
|
||||
if (i >= 0) this.streamSources.splice(i, 1)
|
||||
}
|
||||
nextTime += audioBuf.duration
|
||||
lastEnd = nextTime
|
||||
}
|
||||
} finally {
|
||||
try { reader.releaseLock() } catch { /* */ }
|
||||
}
|
||||
// bis zum Ende der letzten geplanten Charge warten (sofern nicht abgebrochen)
|
||||
if (!this.streamCancelled) {
|
||||
const waitMs = Math.max(0, (lastEnd - this.ctx.currentTime) * 1000)
|
||||
await new Promise((r) => setTimeout(r, waitMs + 60))
|
||||
}
|
||||
this.onSpeaking?.(false)
|
||||
this.stopMeter()
|
||||
}
|
||||
|
||||
private async playNext(): Promise<void> {
|
||||
const buf = this.queue.shift()
|
||||
if (!buf) { this.playing = false; this.stopMeter(); this.onSpeaking?.(false); return }
|
||||
this.playing = true
|
||||
this.onSpeaking?.(true)
|
||||
if (this.ctx.state === "suspended") { try { await this.ctx.resume() } catch { /* */ } }
|
||||
let audioBuf: AudioBuffer
|
||||
try { audioBuf = await this.ctx.decodeAudioData(buf.slice(0)) } catch { return this.playNext() }
|
||||
const src = this.ctx.createBufferSource()
|
||||
src.buffer = audioBuf
|
||||
src.connect(this.analyser)
|
||||
src.onended = () => { void this.playNext() }
|
||||
src.start()
|
||||
this.startMeter()
|
||||
}
|
||||
|
||||
private startMeter() {
|
||||
cancelAnimationFrame(this.raf)
|
||||
const tick = () => {
|
||||
this.analyser.getByteFrequencyData(this.freq as any)
|
||||
const n = Math.min(this.freq.length, 48)
|
||||
let sum = 0, wsum = 0
|
||||
for (let i = 2; i < n; i++) { const m = this.freq[i]; sum += m; wsum += m * i }
|
||||
const avg = sum / (n - 2) / 255
|
||||
const open = Math.min(1, avg * 1.9)
|
||||
this.level.current = open
|
||||
// Spektraler Schwerpunkt 0..1 (wo sitzt die Klang-Energie) -> grobe Vokal-Form
|
||||
const c = sum > 0 ? wsum / sum : 2
|
||||
const cN = Math.max(0, Math.min(1, (c - 2) / (n - 2)))
|
||||
const bright = smoothstep(0.42, 0.72, cN) // hell -> i/e (breiter Mund)
|
||||
const dark = 1 - smoothstep(0.22, 0.5, cN) // dunkel -> o/u (runder Mund)
|
||||
const ih = open * bright
|
||||
const ou = open * dark
|
||||
const aa = open * (1 - Math.max(bright, dark) * 0.85) // sonst offenes "a"
|
||||
// leichte Glaettung gegen Flackern
|
||||
this.level.aa += (aa - this.level.aa) * 0.5
|
||||
this.level.ih += (ih - this.level.ih) * 0.5
|
||||
this.level.ou += (ou - this.level.ou) * 0.5
|
||||
this.raf = requestAnimationFrame(tick)
|
||||
}
|
||||
tick()
|
||||
}
|
||||
|
||||
private stopMeter() {
|
||||
cancelAnimationFrame(this.raf)
|
||||
this.level.current = 0; this.level.aa = 0; this.level.ih = 0; this.level.ou = 0
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
// Perf-Diagnose (Client): Zeit pro Stufe ab „Mikro losgelassen". Opt-in via localStorage lucy_perf=1.
|
||||
// NUR Console (kein POST an den pocket_server — der generiert Audio; POSTs mittendrin
|
||||
// verursachten hakelige Sprachausgabe). (Aus useVoiceAgent extrahiert, Review P2-13a.)
|
||||
|
||||
export const PERF = typeof localStorage !== "undefined" && localStorage.getItem("lucy_perf") === "1"
|
||||
|
||||
export function plogSend(msg: string) { if (PERF) console.log(`[lucy-perf] ${msg}`) }
|
||||
export function plog(label: string, ms: number) { plogSend(`${label} = ${Math.round(ms)}ms`) }
|
||||
@@ -0,0 +1,14 @@
|
||||
// Leichtgewichtige Stimmungs-Heuristik → treibt Lucys Mimik. Regelbasiert (kein Modell).
|
||||
export type Emotion = "neutral" | "happy" | "angry" | "sad" | "surprised" | "relaxed"
|
||||
|
||||
const RULES: [Emotion, RegExp][] = [
|
||||
["happy", /(super|toll|klasse|freu|cool|prima|perfekt|danke|großartig|wunderbar|gerne|haha)/i],
|
||||
["surprised", /(wow|wirklich\?|krass|unglaublich|echt\?|tatsächlich|\?!|!\?|oha)/i],
|
||||
["angry", /(fehler|kaputt|mist|verdammt|nervt|schlecht|problem|ärgerlich|leider nicht|geht nicht)/i],
|
||||
["sad", /(leider|schade|traurig|tut mir leid|entschuldigung|sorry|bedauere)/i],
|
||||
]
|
||||
|
||||
export function sentimentToEmotion(text: string): Emotion {
|
||||
for (const [emo, rx] of RULES) if (rx.test(text)) return emo
|
||||
return "neutral"
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
// Pure Text-Verarbeitung für den Voice-Loop (aus useVoiceAgent extrahiert, Review P2-13a).
|
||||
// Alles hier ist zustandslos und einzeln testbar — der Hook orchestriert nur noch.
|
||||
|
||||
// Hermes schreibt manchmal ASCII-Umlaute (ue/ae/oe/ss) statt ä/ö/ü/ß -> pocket-tts liest die falsch vor.
|
||||
// Gezielt häufige UMLAUT-Stämme zurückwandeln. KONSERVATIV: nur Muster, bei denen ASCII fast immer ein
|
||||
// Umlaut ist (echte "ue"-Wörter wie aktuell/neue/Quelle bleiben unangetastet — die sind hier NICHT gelistet).
|
||||
const UMLAUT_FIX: [RegExp, string][] = [
|
||||
[/ueber/gi, "über"], [/\bfuer\b/gi, "für"], [/natuerlich/gi, "natürlich"], [/zurueck/gi, "zurück"],
|
||||
[/unterstuetz/gi, "unterstütz"], [/ueberpruef/gi, "überprüf"], [/\bpruef/gi, "prüf"], [/gefuehl/gi, "gefühl"],
|
||||
[/\bfuehl/gi, "fühl"], [/\bfuehr/gi, "führ"], [/\bfuenf/gi, "fünf"], [/\bgruen/gi, "grün"], [/\bfrueh/gi, "früh"],
|
||||
[/\bglueck/gi, "glück"], [/\bstueck/gi, "stück"], [/gemuetlich/gi, "gemütlich"], [/verfueg/gi, "verfüg"],
|
||||
[/\bmoecht/gi, "möcht"], [/\bkoenn/gi, "könn"], [/\bwuerd/gi, "würd"], [/\bmuess/gi, "müss"], [/\bduerf/gi, "dürf"],
|
||||
[/\bhaett/gi, "hätt"], [/\bwaer/gi, "wär"], [/\bspaet/gi, "spät"],
|
||||
[/\btaeglich/gi, "täglich"], [/\bnaechst/gi, "nächst"], [/\baehnlich/gi, "ähnlich"], [/waehrend/gi, "während"],
|
||||
[/\bwaehl/gi, "wähl"], [/erklaer/gi, "erklär"], [/\bschoen/gi, "schön"], [/\bgroess/gi, "größ"], [/\bhoer/gi, "hör"],
|
||||
[/\boeffn/gi, "öffn"], [/\bboese/gi, "böse"], [/\bloesch/gi, "lösch"], [/\bstoer/gi, "stör"], [/koennt/gi, "könnt"],
|
||||
// erweitert (30.06.): weitere häufige Stämme + sicheres -ität-Suffix. KONSERVATIV gewählt;
|
||||
// der TTS-Server (_fix_umlauts) hat zusätzlich ein Netz. neu/aktuell/Steuer/Feuer bleiben unberührt.
|
||||
[/itaet/gi, "ität"],
|
||||
[/\bmoeglich/gi, "möglich"], [/gespraech/gi, "gespräch"], [/\bmaerz/gi, "märz"],
|
||||
[/geschaeft/gi, "geschäft"], [/gefaehr/gi, "gefähr"], [/\bhaeng/gi, "häng"], [/\blaeng/gi, "läng"],
|
||||
[/\bmaenner/gi, "männer"], [/\bmaedchen/gi, "mädchen"], [/\bvaeter/gi, "väter"], [/\btraeum/gi, "träum"],
|
||||
[/\bsaetz/gi, "sätz"], [/\bplaetz/gi, "plätz"], [/\bkaelte/gi, "kälte"], [/\bzaehl/gi, "zähl"],
|
||||
[/\baerg/gi, "ärg"], [/\baerzt/gi, "ärzt"], [/aeusser/gi, "äußer"],
|
||||
[/\btuer/gi, "tür"], [/\bkueche/gi, "küche"], [/buech/gi, "büch"], [/\bbuero/gi, "büro"], [/\bbuerg/gi, "bürg"],
|
||||
[/\bsued/gi, "süd"], [/schueler/gi, "schüler"], [/uebung/gi, "übung"], [/kuenstl/gi, "künstl"],
|
||||
[/\bmuede/gi, "müde"], [/\bmuell/gi, "müll"], [/\bdrueck/gi, "drück"], [/stuetz/gi, "stütz"],
|
||||
[/\bmoebel/gi, "möbel"], [/loes/gi, "lös"], [/voellig/gi, "völlig"], [/zwoelf/gi, "zwölf"],
|
||||
[/\bkoenig/gi, "könig"], [/\bhoeh/gi, "höh"], [/\bdoerf/gi, "dörf"], [/\bwoert/gi, "wört"],
|
||||
]
|
||||
export function restoreUmlauts(s: string): string {
|
||||
let out = s
|
||||
for (const [re, rep] of UMLAUT_FIX) out = out.replace(re, rep as string)
|
||||
return out
|
||||
}
|
||||
|
||||
// Optionaler Stimmungs-Tag, den Hermes ans Ende haengen kann (<emo:happy> ...). Steuert NUR den
|
||||
// Gesichtsausdruck -> wird vor Anzeige UND vor dem Vorlesen entfernt. Fallback = Sentiment-Heuristik.
|
||||
export const EMO_TAG = /<emo:(happy|sad|angry|surprised|relaxed|neutral)>/i
|
||||
export function stripEmoTag(s: string): string { return s.replace(/<emo:\w+>/gi, "").trimEnd() }
|
||||
|
||||
// Hermes hängt bei selbst erzeugten Medien (z.B. eigener Screenshot via pc-control) einen Roh-Token
|
||||
// „MEDIA:<datei>" an. Das ist weder Sprech- noch Anzeigetext -> überall entfernen (bis Medien echt
|
||||
// gerendert werden). Betrifft auch Lucys eigene Bildschirm-Sicht, die den Screenshot ohnehin schon liefert.
|
||||
export function stripMedia(s: string): string {
|
||||
return s.replace(/\bMEDIA:\S+/gi, "").replace(/[ \t]{2,}/g, " ").replace(/[ \t]+\n/g, "\n").trimEnd()
|
||||
}
|
||||
|
||||
export function cleanForTTS(s: string): string {
|
||||
return restoreUmlauts(s)
|
||||
.replace(/```[\s\S]*?```/g, " ").replace(/`([^`]*)`/g, "$1")
|
||||
.replace(/\[([^\]]+)\]\([^)]+\)/g, "$1")
|
||||
.replace(/\bMEDIA:\S+/gi, " ")
|
||||
// Zeitzonen-Codes (UTC/CET/GMT…) + evtl. angehängte Zeit: Pocket mangelt die Akronyme zu
|
||||
// Kauderwelsch -> aus der Stimme entfernen (Anzeige behält sie). Lucy sagt die lokale Zeit ohnehin.
|
||||
.replace(/\b(?:UTC|GMT|CET|CEST|MEZ|MESZ|PST|PDT|EST|EDT)\b\s*[+\-]?\d{0,2}(?::\d{2})?/gi, " ")
|
||||
// Zahlen/Uhrzeiten (18:01 -> „achtzehn Uhr eins", 2026 -> „zweitausend…") normalisiert jetzt
|
||||
// der lokale pocket_server (text_norm.py, num2words) — kontextsicher (Modellnummern bleiben).
|
||||
.replace(/https?:\/\/\S+/gi, " ").replace(/www\.\S+/gi, " ").replace(/\b\S+@\S+\.\S+\b/g, " ")
|
||||
.replace(/[*_#>~|`]+/g, " ").replace(/^\s*[-•·]\s+/gm, " ")
|
||||
.replace(/\s*&\s*/g, " und ")
|
||||
.replace(/(\d)\s*%/g, "$1 Prozent").replace(/%/g, " Prozent ")
|
||||
.replace(/(\d)\s*°\s*C?/g, "$1 Grad").replace(/°/g, " Grad ")
|
||||
.replace(/\s*=\s*/g, " gleich ").replace(/\s*\/\s*/g, " ")
|
||||
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}]/gu, "")
|
||||
.replace(/\s+/g, " ").trim()
|
||||
}
|
||||
|
||||
// FRISCHE Session pro App-Start (NICHT in localStorage persistieren). Sonst wächst der Hermes-Verlauf
|
||||
// über alle Starts hinweg unbegrenzt (gesehen: 97k+ Tokens) -> Degeneration/Wiederhol-Schleifen.
|
||||
// Dauerhaftes Gedächtnis liegt ohnehin in Mem0 (sessionunabhängig) -> Continuity bleibt erhalten.
|
||||
export function newSessionId(): string {
|
||||
return "lucy-" + Math.random().toString(36).slice(2) + Date.now().toString(36)
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
import { useCallback, useEffect, useRef, useState } from "react"
|
||||
|
||||
// Tanzen zur Musik (MateEngine-Idee): nimmt den SYSTEM-Ton per Loopback ab (getDisplayMedia, der
|
||||
// Main-Prozess liefert audio:"loopback") und misst die Bass-Energie. Daraus speist sich danceLevel,
|
||||
// das der Avatar in eine rhythmische Tanzbewegung umsetzt. Kein Video wird verwendet (Track sofort gestoppt).
|
||||
export function useDanceAudio() {
|
||||
const level = useRef(0)
|
||||
const [dancing, setDancing] = useState(false)
|
||||
const ctxRef = useRef<AudioContext | null>(null)
|
||||
const streamRef = useRef<MediaStream | null>(null)
|
||||
const rafRef = useRef(0)
|
||||
|
||||
const stop = useCallback(() => {
|
||||
cancelAnimationFrame(rafRef.current)
|
||||
streamRef.current?.getTracks().forEach((t) => { try { t.stop() } catch { /* */ } })
|
||||
streamRef.current = null
|
||||
ctxRef.current?.close().catch(() => {}); ctxRef.current = null
|
||||
level.current = 0; setDancing(false)
|
||||
}, [])
|
||||
|
||||
const start = useCallback(async () => {
|
||||
try {
|
||||
// Video wird vom Handler verlangt, brauchen wir aber nicht -> sofort stoppen, nur Audio behalten
|
||||
const stream = await navigator.mediaDevices.getDisplayMedia({ video: true, audio: true })
|
||||
stream.getVideoTracks().forEach((t) => t.stop())
|
||||
if (stream.getAudioTracks().length === 0) { stream.getTracks().forEach((t) => t.stop()); throw new Error("kein System-Audio") }
|
||||
const ctx = new (window.AudioContext || (window as any).webkitAudioContext)()
|
||||
const src = ctx.createMediaStreamSource(stream)
|
||||
const an = ctx.createAnalyser(); an.fftSize = 256; an.smoothingTimeConstant = 0.7
|
||||
src.connect(an)
|
||||
const freq = new Uint8Array(an.frequencyBinCount)
|
||||
const tick = () => {
|
||||
an.getByteFrequencyData(freq)
|
||||
let s = 0; for (let i = 1; i < 10; i++) s += freq[i] // Bass-/Kick-Bereich
|
||||
level.current = Math.min(1, (s / 9 / 255) * 1.7)
|
||||
rafRef.current = requestAnimationFrame(tick)
|
||||
}
|
||||
tick()
|
||||
ctxRef.current = ctx; streamRef.current = stream
|
||||
setDancing(true)
|
||||
} catch (e) { console.error("Tanz-Audio:", e); stop() }
|
||||
}, [stop])
|
||||
|
||||
const toggleDance = useCallback(() => { dancing ? stop() : void start() }, [dancing, start, stop])
|
||||
|
||||
useEffect(() => () => stop(), [stop])
|
||||
return { danceLevel: level, dancing, toggleDance }
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
import { useCallback, useEffect, useRef, useState } from "react"
|
||||
|
||||
// Push-to-talk-Aufnahme via MediaRecorder. start beim Druecken, stop beim Loslassen -> Audio-Blob an onAudio.
|
||||
export function usePushToTalk(onAudio: (blob: Blob) => void) {
|
||||
const [recording, setRecording] = useState(false)
|
||||
const recRef = useRef<MediaRecorder | null>(null)
|
||||
const chunksRef = useRef<Blob[]>([])
|
||||
const streamRef = useRef<MediaStream | null>(null)
|
||||
|
||||
const start = useCallback(async () => {
|
||||
if (recRef.current) return
|
||||
let stream: MediaStream
|
||||
try {
|
||||
stream = await navigator.mediaDevices.getUserMedia({ audio: true })
|
||||
} catch (e) {
|
||||
console.error("Mikrofon-Zugriff verweigert:", e)
|
||||
return
|
||||
}
|
||||
streamRef.current = stream
|
||||
const mime = MediaRecorder.isTypeSupported("audio/webm;codecs=opus") ? "audio/webm;codecs=opus" : "audio/webm"
|
||||
const rec = new MediaRecorder(stream, { mimeType: mime })
|
||||
chunksRef.current = []
|
||||
rec.ondataavailable = (e) => { if (e.data.size) chunksRef.current.push(e.data) }
|
||||
rec.onstop = () => {
|
||||
const blob = new Blob(chunksRef.current, { type: mime })
|
||||
streamRef.current?.getTracks().forEach((t) => t.stop())
|
||||
streamRef.current = null
|
||||
recRef.current = null
|
||||
setRecording(false)
|
||||
if (blob.size > 1200) onAudio(blob)
|
||||
}
|
||||
rec.start()
|
||||
recRef.current = rec
|
||||
setRecording(true)
|
||||
}, [onAudio])
|
||||
|
||||
const stop = useCallback(() => { recRef.current?.stop() }, [])
|
||||
|
||||
useEffect(() => () => {
|
||||
recRef.current?.stop()
|
||||
streamRef.current?.getTracks().forEach((t) => t.stop())
|
||||
}, [])
|
||||
|
||||
return { recording, start, stop }
|
||||
}
|
||||
@@ -0,0 +1,145 @@
|
||||
import { useEffect, useRef } from "react"
|
||||
import { MicVAD } from "@ricky0123/vad-web"
|
||||
import { BOX_URL } from "../../config"
|
||||
|
||||
// Freisprech-VAD: lauscht dauerhaft am Mikro und liefert komplette Äußerungen als Blob.
|
||||
// Seit Review P1-7 (2026-07-02) Silero VAD v5 (neuronal, via vad-web/onnxruntime-wasm) statt
|
||||
// des RMS-Eigenbaus: erkennt Sprache statt Lautstärke (robust gegen Tastatur/Lüfter) und
|
||||
// erlaubt dadurch ein kürzeres Turn-Ende (450 ms statt 900 ms Stille) ohne Abschneiden.
|
||||
// RÜCKKOPPLUNGS-SCHUTZ bleibt: `paused` (Lucy denkt/spricht) verwirft laufende Erkennung,
|
||||
// + echoCancellation + Abkling-Sperre nach Lucys Antwort.
|
||||
|
||||
interface VADOptions {
|
||||
enabled: boolean // VAD-Modus an?
|
||||
paused: boolean // Lucy beschäftigt (thinking/speaking/listening) -> nicht aufnehmen
|
||||
onUtterance: (blob: Blob) => void
|
||||
onListening?: (active: boolean) => void // UI: gerade Sprache am Aufnehmen?
|
||||
}
|
||||
|
||||
const REDEMPTION_MS = 450 // so lange Stille -> Äußerung zu Ende (war 900 ms beim RMS-VAD)
|
||||
const PRE_SPEECH_PAD_MS = 320 // Vorlauf mitschneiden (erster Wortanfang nicht abschneiden)
|
||||
const MIN_SPEECH_MS = 160 // kürzer = Klick/Räuspern -> verwerfen
|
||||
const COOLDOWN_MS = 450 // nach Lucys Antwort kurz taub (Echo/Lautsprecher abklingen lassen)
|
||||
// Semantische Turn-Detection (Smart Turn v3 auf der Box): meldet sie 'incomplete' (User denkt
|
||||
// mitten im Satz nach), warten wir bis zu HOLD_MS auf die Fortsetzung und hängen sie an,
|
||||
// statt mitten im Gedanken zu antworten. Hart begrenzt, damit Lucy nie ewig schweigt.
|
||||
const HOLD_MS = 1800
|
||||
const MAX_UTTERANCE_S = 30 // Sicherheitsdeckel fürs Zusammenhängen
|
||||
|
||||
// Float32-Samples (16 kHz mono) -> WAV-Blob (PCM16). Ersetzt den MediaRecorder-webm-Umweg:
|
||||
// die Box muss kein Opus mehr dekodieren, Parakeet/Whisper bekommen direkt sauberes WAV.
|
||||
function toWavBlob(samples: Float32Array, sampleRate = 16000): Blob {
|
||||
const buf = new ArrayBuffer(44 + samples.length * 2)
|
||||
const v = new DataView(buf)
|
||||
const writeStr = (off: number, s: string) => { for (let i = 0; i < s.length; i++) v.setUint8(off + i, s.charCodeAt(i)) }
|
||||
writeStr(0, "RIFF"); v.setUint32(4, 36 + samples.length * 2, true); writeStr(8, "WAVE")
|
||||
writeStr(12, "fmt "); v.setUint32(16, 16, true); v.setUint16(20, 1, true); v.setUint16(22, 1, true)
|
||||
v.setUint32(24, sampleRate, true); v.setUint32(28, sampleRate * 2, true); v.setUint16(32, 2, true); v.setUint16(34, 16, true)
|
||||
writeStr(36, "data"); v.setUint32(40, samples.length * 2, true)
|
||||
let off = 44
|
||||
for (let i = 0; i < samples.length; i++, off += 2) {
|
||||
const s = Math.max(-1, Math.min(1, samples[i]))
|
||||
v.setInt16(off, s < 0 ? s * 0x8000 : s * 0x7fff, true)
|
||||
}
|
||||
return new Blob([buf], { type: "audio/wav" })
|
||||
}
|
||||
|
||||
// Fragt die Box, ob die Äußerung semantisch fertig ist. Fehler => true (nie blockieren).
|
||||
async function isTurnComplete(audio: Float32Array): Promise<boolean> {
|
||||
try {
|
||||
const fd = new FormData()
|
||||
fd.append("audio", toWavBlob(audio.subarray(Math.max(0, audio.length - 8 * 16000))), "rec.wav")
|
||||
const r = await fetch(`${BOX_URL}/api/voice/turn`, { method: "POST", body: fd })
|
||||
if (!r.ok) return true
|
||||
return (await r.json()).complete !== false
|
||||
} catch {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
function concatAudio(a: Float32Array, b: Float32Array): Float32Array {
|
||||
const out = new Float32Array(a.length + b.length)
|
||||
out.set(a); out.set(b, a.length)
|
||||
return out
|
||||
}
|
||||
|
||||
export function useVAD({ enabled, paused, onUtterance, onListening }: VADOptions) {
|
||||
const pausedRef = useRef(paused); pausedRef.current = paused
|
||||
const onUtt = useRef(onUtterance); onUtt.current = onUtterance
|
||||
const onLst = useRef(onListening); onLst.current = onListening
|
||||
const resumeAt = useRef(0)
|
||||
|
||||
// Abkling-Sperre: sobald Lucy fertig ist (paused true->false), kurz nicht lauschen
|
||||
useEffect(() => { if (!paused) resumeAt.current = performance.now() + COOLDOWN_MS }, [paused])
|
||||
|
||||
useEffect(() => {
|
||||
if (!enabled) return
|
||||
let cancelled = false
|
||||
let vad: Awaited<ReturnType<typeof MicVAD.new>> | null = null
|
||||
// Semantik-Hold: bei 'incomplete' gepufferte Äußerung, auf die die Fortsetzung wartet.
|
||||
let pending: Float32Array | null = null
|
||||
let holdTimer: ReturnType<typeof setTimeout> | null = null
|
||||
|
||||
const emit = (audio: Float32Array) => {
|
||||
pending = null
|
||||
if (holdTimer) { clearTimeout(holdTimer); holdTimer = null }
|
||||
onUtt.current(toWavBlob(audio))
|
||||
}
|
||||
|
||||
;(async () => {
|
||||
try {
|
||||
vad = await MicVAD.new({
|
||||
model: "v5",
|
||||
// Assets self-hosted (electron.vite.config.ts kopiert sie nach /vad/) — offlinefähig.
|
||||
baseAssetPath: "/vad/",
|
||||
onnxWASMBasePath: "/vad/",
|
||||
positiveSpeechThreshold: 0.5,
|
||||
negativeSpeechThreshold: 0.35,
|
||||
redemptionMs: REDEMPTION_MS,
|
||||
preSpeechPadMs: PRE_SPEECH_PAD_MS,
|
||||
minSpeechMs: MIN_SPEECH_MS,
|
||||
getStream: () => navigator.mediaDevices.getUserMedia({
|
||||
audio: { echoCancellation: true, noiseSuppression: true, autoGainControl: true },
|
||||
}),
|
||||
onSpeechStart: () => {
|
||||
if (pausedRef.current || performance.now() < resumeAt.current) return
|
||||
// User spricht weiter, während eine 'incomplete'-Äußerung gehalten wird ->
|
||||
// Timer stoppen; die Fortsetzung wird in onSpeechEnd angehängt.
|
||||
if (holdTimer) { clearTimeout(holdTimer); holdTimer = null }
|
||||
onLst.current?.(true)
|
||||
},
|
||||
onSpeechEnd: async (audio: Float32Array) => {
|
||||
onLst.current?.(false)
|
||||
// Während Lucy denkt/spricht (oder direkt danach) erkannte Sprache = ihr eigenes
|
||||
// Echo bzw. Nachhall -> verwerfen statt transkribieren.
|
||||
if (cancelled || pausedRef.current || performance.now() < resumeAt.current) { pending = null; return }
|
||||
if (!pending && audio.length < MIN_SPEECH_MS * 16) return // 16 Samples/ms @16 kHz
|
||||
let combined = pending ? concatAudio(pending, audio) : audio
|
||||
if (combined.length > MAX_UTTERANCE_S * 16000) {
|
||||
combined = combined.subarray(combined.length - MAX_UTTERANCE_S * 16000)
|
||||
}
|
||||
const complete = await isTurnComplete(combined)
|
||||
if (cancelled) return
|
||||
if (complete) { emit(combined); return }
|
||||
// Mitten im Gedanken pausiert: kurz auf die Fortsetzung warten, dann notfalls doch senden.
|
||||
pending = combined
|
||||
if (holdTimer) clearTimeout(holdTimer)
|
||||
holdTimer = setTimeout(() => { if (!cancelled && pending) emit(pending) }, HOLD_MS)
|
||||
},
|
||||
onVADMisfire: () => onLst.current?.(false),
|
||||
})
|
||||
if (cancelled) { vad.destroy(); return }
|
||||
vad.start()
|
||||
} catch (e) {
|
||||
console.error("VAD: Silero-Init fehlgeschlagen", e)
|
||||
}
|
||||
})()
|
||||
|
||||
return () => {
|
||||
cancelled = true
|
||||
if (holdTimer) clearTimeout(holdTimer)
|
||||
try { vad?.destroy() } catch { /* */ }
|
||||
onLst.current?.(false)
|
||||
}
|
||||
}, [enabled])
|
||||
}
|
||||
@@ -0,0 +1,305 @@
|
||||
import { useCallback, useEffect, useRef, useState } from "react"
|
||||
import { usePushToTalk } from "./usePushToTalk"
|
||||
import { useVAD } from "./useVAD"
|
||||
import { AudioQueue } from "./audio"
|
||||
import { sentimentToEmotion, type Emotion } from "./sentiment"
|
||||
import { BOX_URL, SYSTEM_PROMPT } from "../../config"
|
||||
import { stt, tts, SpeechScheduler } from "../../voice-core"
|
||||
import { EMO_TAG, cleanForTTS, newSessionId, restoreUmlauts, stripEmoTag, stripMedia } from "./textPipeline"
|
||||
import { VISION_RX, extractWindowName } from "./visionIntent"
|
||||
import { PERF, plog, plogSend } from "./perf"
|
||||
|
||||
// Voll-Duplex-Schleife: PTT/VAD -> Box /stt -> Box /chat (SSE Hermes) -> CHUNKS -> lokal /tts -> AudioQueue.
|
||||
// Hirn+STT = Box (wie WebUI), Stimme = lokal (pocket-tts, CPU). Antwort wird in groessere Bloecke
|
||||
// gebuendelt und satzweise frueh gesprochen. Warm-Gate beim Start: Bedienung erst frei, wenn der
|
||||
// TTS-Dienst antwortet. Text-Verarbeitung: textPipeline.ts · Vision-Intent: visionIntent.ts ·
|
||||
// Perf-Logging: perf.ts (Review P2-13a: Hook = nur noch Orchestrierung).
|
||||
|
||||
export type VoiceStatus = "warming" | "idle" | "listening" | "transcribing" | "thinking" | "speaking" | "error"
|
||||
export interface ChatMsg { role: "user" | "assistant"; text: string }
|
||||
|
||||
export function useVoiceAgent() {
|
||||
const [status, setStatus] = useState<VoiceStatus>("warming")
|
||||
const [ready, setReady] = useState(false)
|
||||
const [messages, setMessages] = useState<ChatMsg[]>([])
|
||||
const [error, setError] = useState<string | null>(null)
|
||||
const [inputMode, setInputModeState] = useState<"ptt" | "vad">(
|
||||
() => (localStorage.getItem("lucy_input_mode") === "vad" ? "vad" : "ptt"))
|
||||
const [vadListening, setVadListening] = useState(false)
|
||||
const setInputMode = useCallback((m: "ptt" | "vad") => {
|
||||
localStorage.setItem("lucy_input_mode", m); setInputModeState(m)
|
||||
}, [])
|
||||
|
||||
const audioLevel = useRef({ current: 0 })
|
||||
const emotion = useRef<Emotion>("neutral")
|
||||
const queueRef = useRef<AudioQueue | null>(null)
|
||||
const schedulerRef = useRef<SpeechScheduler | null>(null)
|
||||
const turnAbortRef = useRef<AbortController | null>(null) // laufenden Turn abbrechen (Barge-in)
|
||||
const thinkingRef = useRef(false) // Hermes werkelt noch (Tools) -> Status nach Filler zurück auf 'thinking'
|
||||
const turnT0 = useRef(0) // Perf: Startzeit des aktuellen Turns (Mikro losgelassen)
|
||||
const sessionId = useRef<string>(newSessionId())
|
||||
// Live-Status fuer die Eingabe-Sperre (ohne pressStart-Closure neu zu binden)
|
||||
const statusRef = useRef<VoiceStatus>("warming")
|
||||
useEffect(() => { statusRef.current = status }, [status])
|
||||
|
||||
const ensureQueue = useCallback(() => {
|
||||
if (!queueRef.current) {
|
||||
const q = new AudioQueue()
|
||||
queueRef.current = q
|
||||
audioLevel.current = q.level
|
||||
// Satz-Pipelining: der Scheduler spricht einzelne Sätze, sobald sie aus dem Hirn-Stream
|
||||
// fertig sind. Er (nicht die Queue) treibt den „speaking"-Status, damit es zwischen Sätzen
|
||||
// nicht flackert (die Queue pausiert die Pegelmessung je Satz).
|
||||
const sch = new SpeechScheduler(tts, q)
|
||||
// Nach dem Filler-Satz zurück auf 'thinking', solange Hermes noch werkelt (sonst flackert's auf idle).
|
||||
sch.onBusy = (busy) => setStatus((s) => (busy ? "speaking" : thinkingRef.current ? "thinking" : s === "speaking" ? "idle" : s))
|
||||
schedulerRef.current = sch
|
||||
}
|
||||
return queueRef.current
|
||||
}, [])
|
||||
|
||||
// --- Warm-Gate: warten bis der lokale TTS-Dienst geladen ist, bevor die Bedienung frei wird ---
|
||||
// pocket_server serviert /health erst, wenn das Modell fertig geladen ist (FastAPI-lifespan).
|
||||
// Der Main-Prozess spawnt den Server beim App-Start; hier nur pollen.
|
||||
useEffect(() => {
|
||||
let cancelled = false
|
||||
;(async () => {
|
||||
// Geduldiger Retry mit Backoff statt einmaligem 120s-Warten: crasht der pocket_server
|
||||
// (oder braucht er länger), bleibt die App in 'warming' MIT sichtbarer Meldung — vorher
|
||||
// schaltete sie nach dem Timeout stumm auf 'ready' und die Stimme fehlte einfach.
|
||||
let attempt = 0
|
||||
while (!cancelled) {
|
||||
const ok = await tts.waitReady(attempt === 0 ? 45_000 : 20_000)
|
||||
if (cancelled) return
|
||||
if (ok) break
|
||||
attempt++
|
||||
setError(`Stimme startet nicht (Versuch ${attempt}) — pocket_server prüfen, ich versuche es weiter …`)
|
||||
await new Promise((r) => setTimeout(r, Math.min(5_000 * attempt, 30_000)))
|
||||
}
|
||||
if (cancelled) return
|
||||
setError(null)
|
||||
// ein Aufwaerm-Satz (primt alle lazy Pfade); Audio verwerfen
|
||||
try { await tts.synthesize("Alles bereit, Commander.") } catch { /* */ }
|
||||
if (cancelled) return
|
||||
setReady(true); setStatus("idle")
|
||||
})()
|
||||
return () => { cancelled = true }
|
||||
}, [])
|
||||
|
||||
// Ein kompletter Turn: User-Text (+ optional Bildschirm-Bilder, Multi-Monitor) -> Hermes (SSE) -> Stimme.
|
||||
const runTurn = useCallback(async (userText: string, images?: string[]) => {
|
||||
const queue = ensureQueue()
|
||||
const scheduler = schedulerRef.current!
|
||||
// Vorherigen Turn (falls noch am Streamen) hart abbrechen -> kein „Weiterreden" nach Barge-in.
|
||||
turnAbortRef.current?.abort()
|
||||
const ac = new AbortController()
|
||||
turnAbortRef.current = ac
|
||||
scheduler.clear(); queue.clear() // frischer Turn: evtl. Reste aus vorherigem Sprechen verwerfen
|
||||
setStatus("thinking")
|
||||
thinkingRef.current = true // Hermes werkelt (evtl. Tools) -> Status nach Filler zurück auf 'thinking'
|
||||
let assistant = ""
|
||||
let firstToken = true
|
||||
setMessages((m) => [...m, { role: "assistant", text: "" }])
|
||||
|
||||
// GANZE Antwort in EINEM Stream an Pocket (nach Hermes-Ende). Pocket kürzt intern den ersten
|
||||
// Chunk (FAST_FIRST) -> schnelles erstes Audio, konsistente Prosodie, wenige Collapse-Regens.
|
||||
// (Client-seitiges Satz-Chunking kämpfte gegen genau diese Optimierung -> verworfen.)
|
||||
let dispatchedAny = false
|
||||
let aborted = false
|
||||
// --- Inkrementelles Sprechen bei TOOL-Turns ------------------------------------------------
|
||||
// Ohne Tools bleibt alles Ein-Stück (Flush erst am Ende) -> gleiche Prosodie/FAST_FIRST wie bisher.
|
||||
// Sobald Hermes ein Tool anstößt (hermes.*-Event), sprechen wir die bis dahin FERTIGEN Sätze schon
|
||||
// -> Lucy redet, WÄHREND das Tool läuft, statt am Ende alles am Stück (sonst 30s+ Totstille bei Tool-Ketten).
|
||||
let spokenLen = 0
|
||||
const SENT_END = /[.!?…](?=[\s"“”„)\]]|$)/g
|
||||
const flushSpeakable = (force: boolean) => {
|
||||
const raw = assistant.slice(spokenLen)
|
||||
if (!raw.trim()) return
|
||||
if (!force && ((raw.match(/```/g)?.length || 0) % 2) === 1) return // offener Code-Zaun -> warten
|
||||
let upto = raw.length
|
||||
if (!force) {
|
||||
let last = -1, m: RegExpExecArray | null
|
||||
SENT_END.lastIndex = 0
|
||||
while ((m = SENT_END.exec(raw))) last = m.index + 1
|
||||
if (last < 0) return // noch kein ganzer Satz fertig -> warten
|
||||
upto = last
|
||||
}
|
||||
spokenLen += upto
|
||||
const seg = cleanForTTS(stripEmoTag(raw.slice(0, upto)))
|
||||
if (seg) { scheduler.push(seg); dispatchedAny = true }
|
||||
}
|
||||
// Sofort-Kontext: lokale Zeit/Datum mitgeben -> Lucy braucht dafür KEIN Tool (spart den 13s-Tool-Tanz).
|
||||
const timeCtx = `\n\n[Sofort-Kontext, DIREKT nutzbar OHNE Tool: Lokale Zeit/Datum beim Commander ist ` +
|
||||
`${new Date().toLocaleString("de-DE", { weekday: "long", day: "numeric", month: "long", year: "numeric", hour: "2-digit", minute: "2-digit" })} Uhr.]`
|
||||
try {
|
||||
const r = await fetch(`${BOX_URL}/api/voice/chat`, {
|
||||
method: "POST", headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ text: userText, session_id: sessionId.current, system: SYSTEM_PROMPT + timeCtx, images: images || [] }),
|
||||
signal: ac.signal,
|
||||
})
|
||||
if (!r.ok || !r.body) throw new Error(`Agent ${r.status}`)
|
||||
const reader = r.body.getReader()
|
||||
const dec = new TextDecoder()
|
||||
let sse = ""
|
||||
for (;;) {
|
||||
if (ac.signal.aborted) break
|
||||
const { done, value } = await reader.read()
|
||||
if (done) break
|
||||
sse += dec.decode(value, { stream: true })
|
||||
const events = sse.split("\n\n"); sse = events.pop() || ""
|
||||
for (const ev of events) {
|
||||
const lines = ev.split("\n")
|
||||
// Hermes mischt Tool-Fortschritt (event: hermes.tool.progress) in den Stream. Das sind KEINE
|
||||
// Chat-Chunks -> nicht als solche parsen (ein Tool-Hickup mit error-Objekt darf die Antwort
|
||||
// nicht abbrechen). Optional koennte man hier Tool-Status anzeigen.
|
||||
const evType = lines.find((l) => l.startsWith("event:"))?.slice(6).trim()
|
||||
if (evType && evType.startsWith("hermes.")) {
|
||||
// Tool-Fortschritt (Status bleibt 'thinking'). Für die „denkt-lange"-Diagnose zeigen wir
|
||||
// die Agent-Aktivität mit Zeitstempel -> so sieht man, ob Tools/Reasoning die Zeit fressen.
|
||||
if (PERF) {
|
||||
const dt = lines.find((l) => l.startsWith("data:"))?.slice(5).trim().slice(0, 140) || ""
|
||||
plogSend(`Hermes ${evType} +${Math.round(performance.now() - turnT0.current)}ms ${dt}`)
|
||||
}
|
||||
flushSpeakable(false) // Tool läuft an -> die bis hier fertigen Sätze schon sprechen (redet WÄHREND das Tool arbeitet)
|
||||
continue
|
||||
}
|
||||
const line = lines.find((l) => l.startsWith("data:"))
|
||||
if (!line) continue
|
||||
const data = line.slice(5).trim()
|
||||
if (data === "[DONE]") continue
|
||||
let json: any
|
||||
try { json = JSON.parse(data) } catch { continue }
|
||||
if (json.error) {
|
||||
const msg = typeof json.error === "string"
|
||||
? json.error
|
||||
: (json.error?.message || JSON.stringify(json.error))
|
||||
throw new Error(msg)
|
||||
}
|
||||
const delta = json.choices?.[0]?.delta?.content || ""
|
||||
if (!delta) continue
|
||||
if (firstToken) { firstToken = false; plog("Chat-TTFB (1. Hermes-Token)", performance.now() - turnT0.current) }
|
||||
assistant += delta
|
||||
// Defensiv: falls Hermes seinen internen Umschlag durchreicht (Mid-Turn) -> NICHTS vorlesen.
|
||||
if (assistant.includes("OUT-OF-BAND USER MESSAGE")) { aborted = true; scheduler.clear(); queue.clear(); break }
|
||||
emotion.current = sentimentToEmotion(assistant)
|
||||
setMessages((m) => { const c = m.slice(); c[c.length - 1] = { role: "assistant", text: stripEmoTag(stripMedia(restoreUmlauts(assistant))) }; return c })
|
||||
}
|
||||
if (aborted) break
|
||||
}
|
||||
if (ac.signal.aborted) return // Turn wurde unterbrochen (Barge-in) -> still beenden
|
||||
if (aborted) {
|
||||
setStatus("error")
|
||||
setError("Nachricht kam mitten im Turn an — bitte warten, bis Lucy fertig ist, dann erneut fragen.")
|
||||
return
|
||||
}
|
||||
if (!assistant.trim()) { setStatus("idle"); return }
|
||||
// Stimmungs-Tag (falls vorhanden) -> Avatar-Mimik nach BEDEUTUNG (sonst bleibt Sentiment-Heuristik).
|
||||
const emo = assistant.match(EMO_TAG)
|
||||
if (emo) emotion.current = emo[1].toLowerCase() as Emotion
|
||||
// Rest sprechen: bei Tool-Turns wurde schon segmentweise geflusht; hier kommt das letzte Segment
|
||||
// (bzw. bei Turns OHNE Tools die GANZE Antwort in einem Rutsch -> FAST_FIRST, gleiche Prosodie wie bisher).
|
||||
thinkingRef.current = false // Hermes fertig -> nach der Antwort darf der Status auf idle (kein Rückfall auf 'thinking')
|
||||
plog("Text an TTS (Rest)", performance.now() - turnT0.current)
|
||||
flushSpeakable(true)
|
||||
await scheduler.idle() // warten, bis Lucy fertig gesprochen hat
|
||||
if (dispatchedAny && !scheduler.playedAny) {
|
||||
setStatus("error")
|
||||
setError("Sprachausgabe fehlgeschlagen — laeuft der lokale TTS-Dienst?")
|
||||
} else if (!dispatchedAny) {
|
||||
setStatus("idle")
|
||||
}
|
||||
} catch (e: any) {
|
||||
if (e?.name === "AbortError" || ac.signal.aborted) return // absichtlich abgebrochen (Barge-in)
|
||||
setStatus("error"); setError(`Agent-Antwort fehlgeschlagen: ${e.message}`)
|
||||
}
|
||||
}, [ensureQueue])
|
||||
|
||||
// Bildschirm-Sicht: Bild(er) erfassen (gezieltes Fenster per Name, sonst ALLE Monitore)
|
||||
const captureForVision = useCallback(async (userText: string): Promise<{ images: string[]; label: string }> => {
|
||||
if (!window.lucy) return { images: [], label: "" }
|
||||
try {
|
||||
const name = extractWindowName(userText)
|
||||
if (name) {
|
||||
const hit = await window.lucy.captureByName(name)
|
||||
if (hit?.image) return { images: [hit.image], label: ` 👁 ${hit.name}` }
|
||||
}
|
||||
const shots = (await window.lucy.captureScreen()).filter(Boolean)
|
||||
if (shots.length) return { images: shots, label: shots.length > 1 ? ` 👁 ${shots.length} Bildschirme` : " 👁 Bildschirm" }
|
||||
} catch (e) { console.error("Bildschirm-Capture:", e) }
|
||||
return { images: [], label: "" }
|
||||
}, [])
|
||||
|
||||
const handleAudio = useCallback(async (blob: Blob) => {
|
||||
setError(null)
|
||||
turnT0.current = performance.now() // Perf: Startpunkt = Mikro losgelassen
|
||||
turnAbortRef.current?.abort() // Barge-in: laufenden Hirn-Turn stoppen (kein Nachschieben)
|
||||
thinkingRef.current = false
|
||||
ensureQueue().clear()
|
||||
schedulerRef.current?.clear() // Barge-in: wartende Sätze mit verwerfen
|
||||
setStatus("transcribing")
|
||||
let userText = ""
|
||||
try {
|
||||
userText = await stt.transcribe(blob)
|
||||
plog("STT (Mikro->Text)", performance.now() - turnT0.current)
|
||||
} catch (e: any) {
|
||||
setStatus("error"); setError(`Spracherkennung fehlgeschlagen: ${e.message}`); return
|
||||
}
|
||||
if (!userText) { setStatus("idle"); return }
|
||||
// Bittet der Nutzer Lucy, auf den Bildschirm zu schauen? -> Bild(er) erfassen + mitschicken.
|
||||
const vis = VISION_RX.test(userText) ? await captureForVision(userText) : { images: [] as string[], label: "" }
|
||||
setMessages((m) => [...m, { role: "user", text: userText + vis.label }])
|
||||
await runTurn(userText, vis.images)
|
||||
}, [ensureQueue, runTurn, captureForVision])
|
||||
|
||||
// Auge-Button: Lucy aktiv auf den Bildschirm schauen lassen (ohne Sprachbefehl)
|
||||
const lookAtScreen = useCallback(async () => {
|
||||
if (!ready || statusRef.current === "thinking" || statusRef.current === "transcribing") return
|
||||
setError(null); turnT0.current = performance.now(); ensureQueue().clear(); schedulerRef.current?.clear()
|
||||
let images: string[] = []
|
||||
try { images = ((await window.lucy?.captureScreen()) || []).filter(Boolean) } catch { /* */ }
|
||||
const prompt = "Schau auf meinen Bildschirm und sag mir kurz, was du darauf siehst."
|
||||
const label = images.length > 1 ? ` 👁 ${images.length} Bildschirme` : " 👁 Bildschirm"
|
||||
setMessages((m) => [...m, { role: "user", text: prompt + label }])
|
||||
await runTurn(prompt, images)
|
||||
}, [ensureQueue, runTurn, ready])
|
||||
|
||||
const { recording, start, stop } = usePushToTalk(handleAudio)
|
||||
|
||||
// Freisprech-VAD: nur im vad-Modus, und nur wenn Lucy IDLE ist (paused sonst) -> keine Rückkopplung,
|
||||
// keine Mid-Turn-Nachricht. ensureQueue() vor handleAudio, damit Barge-in/Audio bereit ist.
|
||||
useVAD({
|
||||
enabled: ready && inputMode === "vad",
|
||||
paused: status !== "idle",
|
||||
onUtterance: (blob) => { ensureQueue(); void handleAudio(blob) },
|
||||
onListening: setVadListening,
|
||||
})
|
||||
|
||||
const pressStart = useCallback(() => {
|
||||
if (!ready) return
|
||||
// Waehrend das Gehirn arbeitet (STT/Hermes-Turn laeuft) KEINEN neuen Turn starten: eine zweite
|
||||
// Nachricht mitten im Turn liefert Hermes als "OUT-OF-BAND" aus -> Modell echot den Umschlag.
|
||||
// Barge-in beim SPRECHEN bleibt erlaubt (Hermes-Turn ist dann fertig) -> queue.clear() in handleAudio.
|
||||
if (statusRef.current === "transcribing" || statusRef.current === "thinking") return
|
||||
ensureQueue()
|
||||
setStatus("listening")
|
||||
void start()
|
||||
}, [ready, ensureQueue, start])
|
||||
|
||||
const pressEnd = useCallback(() => { stop() }, [stop])
|
||||
|
||||
const reset = useCallback(() => {
|
||||
turnAbortRef.current?.abort()
|
||||
thinkingRef.current = false
|
||||
queueRef.current?.clear()
|
||||
schedulerRef.current?.clear()
|
||||
setMessages([]); setError(null); setStatus("idle")
|
||||
sessionId.current = newSessionId() // frischer Gesprächsfaden (Mem0-Gedächtnis bleibt)
|
||||
}, [])
|
||||
|
||||
useEffect(() => {
|
||||
if (!recording && status === "listening") setStatus("transcribing")
|
||||
}, [recording, status])
|
||||
|
||||
return { status, ready, messages, error, recording, audioLevel, emotion, pressStart, pressEnd, reset,
|
||||
inputMode, setInputMode, vadListening, lookAtScreen }
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
// Bildschirm-Sicht-Intent: erkennt im Gesagten die Bitte, auf den Schirm zu schauen
|
||||
// (aus useVoiceAgent extrahiert, Review P2-13a).
|
||||
|
||||
export const VISION_RX = /\b(schau|sieh|siehst|guck|guckst|zeig|bildschirm|screen|monitor|fenster|erkennst?|lies (mir|das)|was (steht|ist) (da|hier|auf)|auf meinem (bildschirm|schirm|screen))\b/i
|
||||
|
||||
export function extractWindowName(text: string): string | null {
|
||||
// grobe Heuristik: Wort/Phrase nach 'fenster|app|programm|in|auf|bei|im'
|
||||
const m = text.match(/(?:fenster|app|programm|in|auf|bei|im)\s+([A-Za-zÄÖÜäöü][\w.+\- ]{2,28})/i)
|
||||
if (!m) return null
|
||||
return m[1].replace(/\b(an|fenster|programm|app|bildschirm|schirm|screen|siehst|du)\b/gi, "").trim() || null
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user