From b14c4ea03f98ca90c1911c402aad770f850c2d55 Mon Sep 17 00:00:00 2001 From: waveringana Date: Wed, 7 Oct 2026 03:00:47 -0400 Subject: [PATCH] valefar: bonsai on the 3060 with a 64K F16 cache, 2 slots, tensor-core attention in batch-invariant mode --- hosts/valefar/bonsai.nix | 21 +++++++----- .../pkgs/llama-cpp-bonsai-bi-mma-f16.patch | 32 +++++++++++++++++++ pkgs-set/pkgs/llama-cpp-bonsai.nix | 4 +++ 3 files changed, 49 insertions(+), 8 deletions(-) create mode 100644 pkgs-set/pkgs/llama-cpp-bonsai-bi-mma-f16.patch diff --git a/hosts/valefar/bonsai.nix b/hosts/valefar/bonsai.nix index 528f4a0..8089eca 100644 --- a/hosts/valefar/bonsai.nix +++ b/hosts/valefar/bonsai.nix @@ -1,6 +1,7 @@ -# Ternary Bonsai 2 27B on the 8GB RTX 4060 through terra.llama-cpp-bonsai (the -# PrismML fork with the PTQ1_0 decode kernel, CUDA sm_89). The lean MTP graft at -# 32K context uses about 7.3 of the 8GB. llama-server stays on loopback; the +# Ternary Bonsai 2 27B on the 12GB RTX 3060 through terra.llama-cpp-bonsai (the +# PrismML fork with the PTQ1_0 decode kernel). The lean MTP graft with a 64K F16 +# cache shared by two slots uses about 11.6 of the 12GB; F16 rather than q4_0 so +# long-context decode takes the tensor-core attention kernel. llama-server stays on loopback; the # socket below is the only thing the LAN and tailnet can reach, and it has no # API key, so both networks are trusted with the GPU. { @@ -18,7 +19,7 @@ in environment.systemPackages = [ terra.llama-cpp-bonsai ]; systemd.services.bonsai2 = { - description = "Ternary Bonsai 2 27B via llama.cpp (MTP, 32K)"; + description = "Ternary Bonsai 2 27B via llama.cpp (MTP, 64K, 2 slots)"; unitConfig.ConditionPathExists = model; after = [ "network-online.target" ]; wants = [ "network-online.target" ]; @@ -41,17 +42,21 @@ in "--alias" "bonsai-2-27b" "-c" - "32768" + "65536" "-ngl" "99" "-fa" "on" "-np" - "1" + "2" + "-kvu" "-ctk" - "q4_0" + "f16" "-ctv" - "q4_0" + "f16" + "-ub" + "1024" + "-bs" "--jinja" "--reasoning-effort" "medium" diff --git a/pkgs-set/pkgs/llama-cpp-bonsai-bi-mma-f16.patch b/pkgs-set/pkgs/llama-cpp-bonsai-bi-mma-f16.patch new file mode 100644 index 0000000..84820b2 --- /dev/null +++ b/pkgs-set/pkgs/llama-cpp-bonsai-bi-mma-f16.patch @@ -0,0 +1,32 @@ +diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu +index b5042840..58620e74 100644 +--- a/ggml/src/ggml-cuda/fattn.cu ++++ b/ggml/src/ggml-cuda/fattn.cu +@@ -10,6 +10,15 @@ static void ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ggml_backend_cuda_con + const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; + const ggml_tensor * Q = dst->src[0]; + ++ // batch-invariant mode: one tile width for 1 to 4 queries, so a token verified in a speculative batch ++ // attends with the same arithmetic as a token decoded alone ++ if constexpr (ncols2 <= 8) { ++ if (ggml_cuda_batch_invariant() && turing_mma_available(cc) && Q->ne[1] <= 4 && Q->ne[1]*ncols2 <= 32) { ++ ggml_cuda_flash_attn_ext_mma_f16_case(ctx, dst); ++ return; ++ } ++ } ++ + if constexpr (ncols2 <= 8) { + if (turing_mma_available(cc) && Q->ne[1] <= 8/ncols2) { + ggml_cuda_flash_attn_ext_mma_f16_case(ctx, dst); +@@ -464,6 +473,11 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const + // speculative batch attends with the same arithmetic as a token decoded alone (the whole-model + // guarantee is 1 to 4 queries, see ggml_cuda_batch_invariant) + if (ggml_cuda_batch_invariant() && Q->ne[1] <= 8 && Q->ne[3] == 1) { ++ // with an F16 cache the tensor-core kernel is much faster at depth (39 vs 54 ms per token at 64K on ++ // an RTX 3060); ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1 keeps it batch-invariant up to 4 queries ++ if (K->type == GGML_TYPE_F16 && V->type == GGML_TYPE_F16 && gqa_opt_applies && Q->ne[1] <= 4) { ++ return BEST_FATTN_KERNEL_MMA_F16; ++ } + return BEST_FATTN_KERNEL_VEC; + } + if (!ggml_is_quantized(K->type) && !ggml_is_quantized(V->type)) { diff --git a/pkgs-set/pkgs/llama-cpp-bonsai.nix b/pkgs-set/pkgs/llama-cpp-bonsai.nix index 6d7b30d..d9c0069 100644 --- a/pkgs-set/pkgs/llama-cpp-bonsai.nix +++ b/pkgs-set/pkgs/llama-cpp-bonsai.nix @@ -29,6 +29,10 @@ cudaPackages.backendStdenv.mkDerivation (finalAttrs: { hash = "sha256-CQOTOY5IdnQA6c6DceD5xYxYddMnJ9SPy8ihbBJhvME="; }; + # Batch-invariant mode sends 1 to 4 query tokens over an F16 cache through one tensor-core tile, + # instead of the vector kernel that slows down at long contexts. + patches = [ ./llama-cpp-bonsai-bi-mma-f16.patch ]; + nativeBuildInputs = [ cmake ninja -- 2.51.2