114 lines
4.8 KiB
Python
114 lines
4.8 KiB
Python
"""HuggingFace-Helfer: GGUF-Dateien eines Repos auflösen (inkl. Split-Teile) + Größen,
|
|
freie Suche, Repo-URL→ID, verfügbare Quants."""
|
|
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
import httpx
|
|
|
|
|
|
def normalize_repo(s: str) -> str:
|
|
"""Akzeptiert volle HF-URL oder `org/repo` → liefert immer `org/repo`."""
|
|
s = (s or "").strip()
|
|
m = re.search(r"huggingface\.co/([^/\s]+/[^/\s?#]+)", s)
|
|
if m:
|
|
return m.group(1)
|
|
return s.strip("/")
|
|
|
|
|
|
def list_quants(repo: str) -> list[dict]:
|
|
"""Verfügbare Quant-Stufen eines Repos mit Downloadgröße (alle Teile + mmproj), ohne mmproj
|
|
als eigene Stufe. Ein Aufruf der Hugging-Face-API für alle Stufen."""
|
|
baum = _tree(repo)
|
|
quants: set[str] = set()
|
|
for e in baum:
|
|
p = str(e.get("path", ""))
|
|
if p.lower().endswith(".gguf") and "mmproj" not in p.lower():
|
|
m = re.search(r"(I?Q\d[\w]*|F16|BF16|FP16|F32)", p, re.IGNORECASE)
|
|
if m:
|
|
quants.add(m.group(1).upper())
|
|
# gängige Reihenfolge zuerst
|
|
order = {"Q4_K_M": 0, "Q4_K_S": 1, "Q5_K_M": 2, "Q6_K": 3, "Q8_0": 4, "Q3_K_M": 5, "Q2_K": 6}
|
|
return [{"quant": q, "total_bytes": auswahl(baum, q)["total_bytes"]}
|
|
for q in sorted(quants, key=lambda q: (order.get(q, 99), q))]
|
|
|
|
|
|
def search(q: str = "", limit: int = 24) -> list[dict]:
|
|
"""Freie HF-Suche nach GGUF-Repos. Ohne q → Top-GGUF nach Downloads (Stöbern)."""
|
|
params: dict[str, str | int] = {"filter": "gguf", "sort": "downloads", "direction": -1, "limit": limit}
|
|
if q and q.strip():
|
|
params["search"] = q.strip() # von httpx kodiert (vorher roh an die URL gehängt)
|
|
try:
|
|
with httpx.Client(timeout=12.0) as c:
|
|
data = c.get("https://huggingface.co/api/models", params=params).json()
|
|
except Exception:
|
|
return []
|
|
out = []
|
|
for m in (data if isinstance(data, list) else []):
|
|
rid = m.get("id")
|
|
if rid:
|
|
out.append({"repo": rid, "downloads": int(m.get("downloads") or 0),
|
|
"likes": int(m.get("likes") or 0)})
|
|
return out
|
|
|
|
|
|
def hf_bin() -> str:
|
|
"""Pfad zur `hf`-CLI (bevorzugt neben dem laufenden Python im venv)."""
|
|
cand = os.path.join(os.path.dirname(sys.executable), "hf")
|
|
return cand if os.path.exists(cand) else "hf"
|
|
|
|
|
|
def _tree(repo: str) -> list[dict]:
|
|
url = f"https://huggingface.co/api/models/{repo}/tree/main?recursive=true"
|
|
with httpx.Client(timeout=20.0) as c:
|
|
data = c.get(url).json()
|
|
return data if isinstance(data, list) else []
|
|
|
|
|
|
def _size(entry: dict) -> int:
|
|
return int(entry.get("size") or (entry.get("lfs") or {}).get("size") or 0)
|
|
|
|
|
|
def resolve_gguf(repo: str, quant: str = "Q4_K_M") -> dict:
|
|
"""Beste GGUF-Auswahl eines Repos für einen Quant (siehe auswahl)."""
|
|
return auswahl(_tree(repo), quant)
|
|
|
|
|
|
def auswahl(tree: list[dict], quant: str = "Q4_K_M") -> dict:
|
|
"""GGUF-Auswahl aus dem Dateibaum eines Repos für einen Quant. Behandelt Split-GGUFs
|
|
(-00001-of-000NN) als Gruppe. Liefert die Datei-/Pattern-Infos für den Download.
|
|
|
|
Rückgabe: {files:[paths], first:path, total_bytes:int, mmproj:path|None, split:bool}
|
|
"""
|
|
ggufs = [e for e in tree if str(e.get("path", "")).lower().endswith(".gguf")]
|
|
q = quant.lower()
|
|
# mmproj separat (Vision-Projektor)
|
|
mmproj = next((e["path"] for e in ggufs if "mmproj" in e["path"].lower()), None)
|
|
model = [e for e in ggufs if "mmproj" not in e["path"].lower()]
|
|
# Nur der gewünschte Quant. Bis 24.09.2026 fiel die Auswahl sonst auf ALLE Modelldateien zurück —
|
|
# bei aufgeteilten Repos lud der Download dann alle Teile aller Varianten (hunderte GB).
|
|
chosen = [e for e in model if q in e["path"].lower()]
|
|
if not chosen:
|
|
return {"files": [], "first": None, "total_bytes": 0, "mmproj": mmproj, "split": False}
|
|
# Split? Wenn die gewählten Dateien -of- enthalten → alle Teile EINER Gruppe (gleicher Präfix).
|
|
split = any("-of-" in e["path"].lower() for e in chosen)
|
|
if split:
|
|
def _gruppe(pfad: str) -> str:
|
|
return re.sub(r"-\d{5}-of-\d{5}\.gguf$", "", pfad, flags=re.IGNORECASE)
|
|
erste = min(e["path"] for e in chosen if "-of-" in e["path"].lower())
|
|
parts = sorted([e for e in chosen if "-of-" in e["path"].lower() and _gruppe(e["path"]) == _gruppe(erste)],
|
|
key=lambda e: e["path"])
|
|
files = [e["path"] for e in parts]
|
|
first = files[0]
|
|
total = sum(_size(e) for e in parts)
|
|
else:
|
|
# ein einzelnes File: nimm das kleinste passende (typisch genau eins)
|
|
chosen.sort(key=lambda e: _size(e))
|
|
first = chosen[0]["path"]
|
|
files = [first]
|
|
total = _size(chosen[0])
|
|
if mmproj:
|
|
total += next((_size(e) for e in ggufs if e["path"] == mmproj), 0)
|
|
return {"files": files, "first": first, "total_bytes": total, "mmproj": mmproj, "split": split}
|