From 431509d14a7c1fbaec000c961b951123ec858523 Mon Sep 17 00:00:00 2001 From: sb32445 Date: Fri, 2 Oct 2026 09:17:12 +0200 Subject: [PATCH 1/3] cuda: use the MMA flash attention kernel for GQA above 4 with quantized K/V on Ada With quantized K/V on Ada the vector kernel was picked for 1 and 2 queries at every context length. At long context it is much slower than the tensor core kernel. The branch for F16 K/V already avoids the vector kernel when the GQA ratio is above 4; this applies the same ratio rule to the quantized branch. Ternary Bonsai 2 27B PTQ1_0 (head size 256, GQA 6, q4_0 K/V, mean-centered), RTX 4070 12 GB, greedy decode, one slot, tokens per second by filled context: depth no MTP: before -> after draft-mtp n-max 1: before -> after 32768 44.8 -> 54.2 53.3 -> 84.0 120000 25.3 -> 41.2 23.2 -> 68.5 draft-mtp n-max 2 (3 queries, already on the MMA kernel) and depth 0 are unchanged. llama-bench pp1/pp2 at depth 2048..16384 shows the MMA kernel is not slower from 2k keys on. test-backend-ops FLASH_ATTN_EXT passes (2994/2994). With 12 extra local cases for q4_0 and q8_0 at GQA 6, head size 256, 1 and 2 queries it passes on both paths (3006/3006). GGML_CUDA_FATTN_GQA_MMA=0 restores the old choice, GGML_CUDA_FATTN_VEC_MAXQ changes the query limit of the vector kernel (default 2). Co-Authored-By: Claude Sonnet 5.5 --- ggml/src/ggml-cuda/fattn.cu | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu index 59e6c9ba74b0..dbc8e603cb81 100644 --- a/ggml/src/ggml-cuda/fattn.cu +++ b/ggml/src/ggml-cuda/fattn.cu @@ -4,6 +4,7 @@ #include "fattn-tile.cuh" #include "fattn-vec.cuh" #include "fattn.cuh" +#include template static void ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { @@ -487,7 +488,12 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const } } else { if (cc >= GGML_CUDA_CC_ADA_LOVELACE) { - if (Q->ne[1] <= 2) { + // The vector kernel loses to the MMA kernel on Ada once the GQA ratio is above 4 (same rule as for + // F16 K/V above): measured on Bonsai 2 27B (GQA 6, head size 256, q4_0 K/V), 1-2 queries at 32k-120k keys. + // GGML_CUDA_FATTN_GQA_MMA=0 restores the old choice, GGML_CUDA_FATTN_VEC_MAXQ changes the query limit. + static const bool gqa_mma = [] { const char * s = getenv("GGML_CUDA_FATTN_GQA_MMA"); return s ? atoi(s) != 0 : true; }(); + static const int vec_max_q = [] { const char * s = getenv("GGML_CUDA_FATTN_VEC_MAXQ"); return s ? atoi(s) : 2; }(); + if (Q->ne[1] <= vec_max_q && !(gqa_mma && gqa_ratio > 4)) { return BEST_FATTN_KERNEL_VEC; } } else { From 069cd48fa8e2b993911092df80f3ed5b1af16ff3 Mon Sep 17 00:00:00 2001 From: sb32445 Date: Sat, 3 Oct 2026 14:57:41 +0200 Subject: [PATCH 2/3] cuda: remove the GQA/MMA and vector-kernel query-limit switches GGML_CUDA_FATTN_GQA_MMA and GGML_CUDA_FATTN_VEC_MAXQ of the previous commit were only there to measure the change. The vector kernel keeps the original limit of 2 queries and is used for GQA <= 4 only. Co-Authored-By: Claude Sonnet 5.5 --- ggml/src/ggml-cuda/fattn.cu | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu index dbc8e603cb81..7a070fb5aba4 100644 --- a/ggml/src/ggml-cuda/fattn.cu +++ b/ggml/src/ggml-cuda/fattn.cu @@ -4,7 +4,6 @@ #include "fattn-tile.cuh" #include "fattn-vec.cuh" #include "fattn.cuh" -#include template static void ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { @@ -490,10 +489,7 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const if (cc >= GGML_CUDA_CC_ADA_LOVELACE) { // The vector kernel loses to the MMA kernel on Ada once the GQA ratio is above 4 (same rule as for // F16 K/V above): measured on Bonsai 2 27B (GQA 6, head size 256, q4_0 K/V), 1-2 queries at 32k-120k keys. - // GGML_CUDA_FATTN_GQA_MMA=0 restores the old choice, GGML_CUDA_FATTN_VEC_MAXQ changes the query limit. - static const bool gqa_mma = [] { const char * s = getenv("GGML_CUDA_FATTN_GQA_MMA"); return s ? atoi(s) != 0 : true; }(); - static const int vec_max_q = [] { const char * s = getenv("GGML_CUDA_FATTN_VEC_MAXQ"); return s ? atoi(s) : 2; }(); - if (Q->ne[1] <= vec_max_q && !(gqa_mma && gqa_ratio > 4)) { + if (Q->ne[1] <= 2 && gqa_ratio <= 4) { return BEST_FATTN_KERNEL_VEC; } } else { From d01570f8a44d03239eb755ba6d1150845d591175 Mon Sep 17 00:00:00 2001 From: sb32445 Date: Wed, 7 Oct 2026 22:43:46 +0200 Subject: [PATCH 3/3] cuda : keep the vector FA kernel when MMA cannot read quantized K/V in place The GQA > 4 route to the MMA kernel on Ada only reads q4_0 and q8_0 K/V in place for head size 128 and 256 with equal K and V types. For other cases the MMA path converts the whole cache to f16 on every call. Check ggml_cuda_fattn_mma_kv_native_supported() before leaving the vector kernel. --- ggml/src/ggml-cuda/fattn.cu | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu index 7a070fb5aba4..787e0fe138ef 100644 --- a/ggml/src/ggml-cuda/fattn.cu +++ b/ggml/src/ggml-cuda/fattn.cu @@ -487,9 +487,9 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const } } else { if (cc >= GGML_CUDA_CC_ADA_LOVELACE) { - // The vector kernel loses to the MMA kernel on Ada once the GQA ratio is above 4 (same rule as for - // F16 K/V above): measured on Bonsai 2 27B (GQA 6, head size 256, q4_0 K/V), 1-2 queries at 32k-120k keys. - if (Q->ne[1] <= 2 && gqa_ratio <= 4) { + // Above GQA 4 the MMA kernel is faster on Ada (same rule as for F16 K/V above), but only when it reads K/V in place. + // Otherwise it converts the whole cache to f16 on every call, so keep the vector kernel. + if (Q->ne[1] <= 2 && (gqa_ratio <= 4 || !ggml_cuda_fattn_mma_kv_native_supported(dst))) { return BEST_FATTN_KERNEL_VEC; } } else {