feat: lower memory dedupe threshold for more aggressive cleaning

This commit is contained in:
root
2026-07-07 16:45:30 +02:00
parent c92f238d2d
commit 97be0f1d00
68 changed files with 15783 additions and 15783 deletions
+60 -60
View File
@@ -1,60 +1,60 @@
{
"_comment": "Kuratierter Modell-Katalog (Cookbook) für Strix Halo / Ryzen AI MAX+ 395 — 128GB unified, bandbreiten-limitiert (256 GB/s). MoE-first. EINE Quelle der Wahrheit für KORREKTE Metadaten (total/active params, moe, generation) → präzise Empfehlungen ohne Namens-Raterei. Inspiriert vom Odysseus-Cookbook (statischer, validierter Katalog statt Live-Scraping). Erweiterbar: neue Modelle hier eintragen. Felder: name (Match-Identifier), repo (HF org/name für Install), family (+Subtyp), generation (numerisch, für Upgrade-Vergleich), total_params_b, active_params_b (=total bei dense), moe, quant, ctx (empfohlen), tools, vision.",
"version": "2026-07-03",
"models": [
{
"role": "fast", "name": "Qwen3.6-35B-A3B", "repo": "Qwen/Qwen3.6-35B-A3B-GGUF",
"family": "qwen", "generation": 3.6, "total_params_b": 35, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": true
},
{
"role": "fast", "name": "Qwen3-30B-A3B-Instruct", "repo": "unsloth/Qwen3-30B-A3B-Instruct-2507-GGUF",
"family": "qwen", "generation": 3.0, "total_params_b": 30, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": false
},
{
"role": "heavy", "name": "Qwen3.5-122B-A10B", "repo": "Qwen/Qwen3.5-122B-A10B-GGUF",
"family": "qwen", "generation": 3.5, "total_params_b": 122, "active_params_b": 10,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": false
},
{
"role": "heavy", "name": "gpt-oss-120b", "repo": "ggml-org/gpt-oss-120b-GGUF",
"family": "gpt-oss", "generation": 1.0, "total_params_b": 120, "active_params_b": 5,
"moe": true, "quant": "MXFP4", "ctx": 32768, "tools": true, "vision": false
},
{
"role": "coder", "name": "Qwen3-Coder-Next", "repo": "Qwen/Qwen3-Coder-Next-GGUF",
"family": "qwen-coder", "generation": 3.0, "total_params_b": 84, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 65536, "tools": true, "vision": false
},
{
"role": "coder", "name": "Qwen3-Coder-30B-A3B-Instruct", "repo": "unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF",
"family": "qwen-coder", "generation": 3.0, "total_params_b": 30, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 65536, "tools": true, "vision": false
},
{
"role": "vision", "name": "Qwen3-VL-30B-A3B-Instruct", "repo": "Qwen/Qwen3-VL-30B-A3B-Instruct-GGUF",
"family": "qwen-vl", "generation": 3.0, "total_params_b": 30, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
},
{
"role": "vision", "name": "Qwen3-VL-8B-Instruct", "repo": "Qwen/Qwen3-VL-8B-Instruct-GGUF",
"family": "qwen-vl", "generation": 3.0, "total_params_b": 8, "active_params_b": 8,
"moe": false, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
},
{
"role": "vision", "name": "Qwen3-VL-2B-Instruct", "repo": "Qwen/Qwen3-VL-2B-Instruct-GGUF",
"family": "qwen-vl", "generation": 3.0, "total_params_b": 2, "active_params_b": 2,
"moe": false, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
},
{
"role": "scout", "name": "GLM-4.6V-Flash", "repo": "ggml-org/GLM-4.6V-Flash-GGUF",
"family": "glm", "generation": 4.6, "total_params_b": 9, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": true
}
]
}
{
"_comment": "Kuratierter Modell-Katalog (Cookbook) für Strix Halo / Ryzen AI MAX+ 395 — 128GB unified, bandbreiten-limitiert (256 GB/s). MoE-first. EINE Quelle der Wahrheit für KORREKTE Metadaten (total/active params, moe, generation) → präzise Empfehlungen ohne Namens-Raterei. Inspiriert vom Odysseus-Cookbook (statischer, validierter Katalog statt Live-Scraping). Erweiterbar: neue Modelle hier eintragen. Felder: name (Match-Identifier), repo (HF org/name für Install), family (+Subtyp), generation (numerisch, für Upgrade-Vergleich), total_params_b, active_params_b (=total bei dense), moe, quant, ctx (empfohlen), tools, vision.",
"version": "2026-07-03",
"models": [
{
"role": "fast", "name": "Qwen3.6-35B-A3B", "repo": "Qwen/Qwen3.6-35B-A3B-GGUF",
"family": "qwen", "generation": 3.6, "total_params_b": 35, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": true
},
{
"role": "fast", "name": "Qwen3-30B-A3B-Instruct", "repo": "unsloth/Qwen3-30B-A3B-Instruct-2507-GGUF",
"family": "qwen", "generation": 3.0, "total_params_b": 30, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": false
},
{
"role": "heavy", "name": "Qwen3.5-122B-A10B", "repo": "Qwen/Qwen3.5-122B-A10B-GGUF",
"family": "qwen", "generation": 3.5, "total_params_b": 122, "active_params_b": 10,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": false
},
{
"role": "heavy", "name": "gpt-oss-120b", "repo": "ggml-org/gpt-oss-120b-GGUF",
"family": "gpt-oss", "generation": 1.0, "total_params_b": 120, "active_params_b": 5,
"moe": true, "quant": "MXFP4", "ctx": 32768, "tools": true, "vision": false
},
{
"role": "coder", "name": "Qwen3-Coder-Next", "repo": "Qwen/Qwen3-Coder-Next-GGUF",
"family": "qwen-coder", "generation": 3.0, "total_params_b": 84, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 65536, "tools": true, "vision": false
},
{
"role": "coder", "name": "Qwen3-Coder-30B-A3B-Instruct", "repo": "unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF",
"family": "qwen-coder", "generation": 3.0, "total_params_b": 30, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 65536, "tools": true, "vision": false
},
{
"role": "vision", "name": "Qwen3-VL-30B-A3B-Instruct", "repo": "Qwen/Qwen3-VL-30B-A3B-Instruct-GGUF",
"family": "qwen-vl", "generation": 3.0, "total_params_b": 30, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
},
{
"role": "vision", "name": "Qwen3-VL-8B-Instruct", "repo": "Qwen/Qwen3-VL-8B-Instruct-GGUF",
"family": "qwen-vl", "generation": 3.0, "total_params_b": 8, "active_params_b": 8,
"moe": false, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
},
{
"role": "vision", "name": "Qwen3-VL-2B-Instruct", "repo": "Qwen/Qwen3-VL-2B-Instruct-GGUF",
"family": "qwen-vl", "generation": 3.0, "total_params_b": 2, "active_params_b": 2,
"moe": false, "quant": "Q4_K_M", "ctx": 32768, "tools": false, "vision": true
},
{
"role": "scout", "name": "GLM-4.6V-Flash", "repo": "ggml-org/GLM-4.6V-Flash-GGUF",
"family": "glm", "generation": 4.6, "total_params_b": 9, "active_params_b": 3,
"moe": true, "quant": "Q4_K_M", "ctx": 32768, "tools": true, "vision": true
}
]
}