diff --git a/llama/models.ini b/llama/models.ini index 2733479..7f3d06f 100644 --- a/llama/models.ini +++ b/llama/models.ini @@ -18,7 +18,6 @@ models-max = 1 ; Qwen3.6-35B-A3B — Alibaba, MoE 3B-active + multimodal. Fast all-rounder. ; --------------------------------------------------------------------------- [qwen3.6-35b-a3b] -load-on-startup = true model = /home/spencer/.cache/models/Qwen3.6-35B-A3B-UD-Q8_K_XL.gguf mmproj = /home/spencer/.cache/models/Qwen3.6-35B-A3B-mmproj-BF16.gguf spec-type = draft-mtp @@ -38,7 +37,7 @@ repeat-penalty = 1.0 ; Qwen3.8-27B — Alibaba, dense + multimodal. Deep reasoning. ; --------------------------------------------------------------------------- [qwen3.8-27b] -model = /home/spencer/.cache/models/Qwen3.8-27B-UD-Q8_K_XL.gguf +model = /home/spencer/.cache/models/Qwen3.8-27B-UD-Q4_K_M.gguf mmproj = /home/spencer/.cache/models/Qwen3.8-27B-mmproj-BF16.gguf spec-type = draft-mtp spec-draft-n-max = 4 @@ -66,3 +65,27 @@ top-p = 0.95 top-k = 40 min-p = 0.01 repeat-penalty = 1.0 + +; --------------------------------------------------------------------------- +; Qwen3.8-Flash-Next — Qwen4-preview arch (qwen4exp), 125B total / 6B active. +; Thinking-mode sampling per official card. No MTP (unsupported yet); vision via +; mmproj-BF16. UD-Q4_K_XL = ~104 GiB, PLE ngram table (~30 GiB) +; lazily served from SSD via mmap → ~74 GiB resident, fits 124 GiB GTT solo. +; Speculative decoding via ngram-mod (long-match proposal). +; --------------------------------------------------------------------------- +[qwen3.8-flash-next] +model = /home/spencer/.cache/models/Qwen3.8-Flash-Next-UD-Q4_K_XL-00001-of-00004.gguf +mmproj = /home/spencer/.cache/models/Qwen3.8-Flash-Next-mmproj-BF16.gguf +chat-template-file = /home/spencer/.config/llama.cpp/qwen-fixed-chat_template.jinja +reasoning-format = deepseek +reasoning = auto +reasoning-preserve = true +spec-type = ngram-mod +spec-ngram-mod-n-max = 32 +spec-ngram-mod-n-min = 16 +temperature = 1.0 +top-p = 0.95 +top-k = 20 +min-p = 0.0 +presence-penalty = 0.0 +repeat-penalty = 1.0 diff --git a/pi/models.json b/pi/models.json index 4b8a370..e0d2154 100644 --- a/pi/models.json +++ b/pi/models.json @@ -76,6 +76,26 @@ "max": null } }, + "qwen3.8-flash-next": { + "name": "Qwen3.8 Flash Next (llama.cpp)", + "reasoning": true, + "input": ["text", "image"], + "contextWindow": 262144, + "maxTokens": 262144, + "compat": { + "thinkingFormat": "reasoning_effort", + "supportsReasoningEffort": true + }, + "thinkingLevelMap": { + "off": "none", + "minimal": null, + "low": "low", + "medium": "medium", + "high": null, + "xhigh": "xhigh", + "max": null + } + }, "qwen3-coder-next-80b-a3b": { "name": "Qwen3 Coder Next 80B A3B (llama.cpp)", "reasoning": false,