version = 1 ; ============================================================================ ; llama.cpp router presets — Strix Halo (Radeon 8060S, 126 GiB unified VRAM) ; ============================================================================ ; --------------------------------------------------------------------------- ; Global defaults shared by every model instance (overridable per-model) ; --------------------------------------------------------------------------- [*] n-gpu-layers = all flash-attn = on jinja = true metrics = true ; Prometheus /metrics endpoint (per-model: /metrics?model=...) ; --------------------------------------------------------------------------- ; Qwen3.6-35B-A3B — Alibaba, MoE 3B-active + multimodal. Fast all-rounder. ; --------------------------------------------------------------------------- [qwen3.6-35b-a3b] load-on-startup = true model = /home/spencer/.cache/models/Qwen3.6-35B-A3B-UD-Q8_K_XL.gguf mmproj = /home/spencer/.cache/models/Qwen3.6-35B-A3B-mmproj-BF16.gguf spec-type = draft-mtp spec-draft-n-max = 3 ubatch-size = 2048 chat-template-file = /home/spencer/.config/llama.cpp/qwen-fixed-chat_template.jinja reasoning-format = deepseek reasoning = auto reasoning-preserve = true temperature = 1.0 top-p = 0.95 top-k = 20 min-p = 0.0 presence-penalty = 1.5 repeat-penalty = 1.0 ; --------------------------------------------------------------------------- ; Qwen3.8-Flash-Next — Qwen4-preview arch (qwen4exp), 125B total / 6B active. ; ; Speculative decoding: DRAFT-MTP (unsloth qwen4exp/mtp fork build provides it; upstream PR ; ggml-org/llama.cpp#27836 still can't load head-only MTP GGUFs). Head = unsloth shared-Q4_K_M. ; n-gram/PLE table is lazy-from-disk in the fork loader (TENSOR_READ_LAZY, no flag needed). ; --------------------------------------------------------------------------- [qwen3.8-flash-next] ; unsloth UD-Q4_K_XL + MTP head (their qwen4exp/mtp fork build — REQUIRED for MTP). ; n-gram/PLE table is served from disk lazily by the fork loader by default (TENSOR_READ_LAZY). ; The shared head excludes embed_tokens (shares the main model's). model = /home/spencer/.cache/models/Qwen3.8-Flash-Next-UD-Q4_K_XL-00001-of-00004.gguf mmproj = /home/spencer/.cache/models/Qwen3.8-Flash-Next-mmproj-BF16.gguf spec-draft-model = /home/spencer/.cache/models/mtp-Qwen3.8-Flash-Next-shared-Q4_K_M.gguf chat-template-file = /home/spencer/.config/llama.cpp/qwen-fixed-chat_template.jinja reasoning-format = deepseek reasoning = auto reasoning-preserve = true spec-type = draft-mtp spec-draft-n-max = 5 temperature = 1.0 top-p = 0.95 top-k = 20 min-p = 0.0 presence-penalty = 0.0 repeat-penalty = 1.0