From 5813a3f0521bd3b34dc397b411489f476993fc79 Mon Sep 17 00:00:00 2001 From: sb32445 Date: Fri, 9 Oct 2026 09:49:24 +0200 Subject: [PATCH] cuda : take the short-query MMA FA shortcut only for the f16 K/V kernel The shortcut added in #189 calls the 64 column MMA kernel without type_KV, so it always builds the f16 variant. With head size 128 or 256 and q4_0/q8_0 K/V the in-place quantized kernel is used and no f16 scratch is reserved after dst. For 3 or 4 queries the shortcut then converted K and V into memory behind the reserved range (invalid write found with compute-sanitizer in test-backend-ops FLASH_ATTN_EXT, hsk=256 nq=3 q4_0, GQA 6). On the MTP decode path this also changes the output and lowers the draft acceptance. The fix adds type_KV == GGML_TYPE_F16 to the condition. AI assistance: this fix was written and checked with Claude Code (AI). It was found with compute-sanitizer, built and tested on an RTX 4070 (test-backend-ops -o FLASH_ATTN_EXT 4007/4007; sanitizer 0 errors on the #189 commit with the fix) and measured with 8 interleaved A/B pairs of MTP decode. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_01Li7aNoBEvKTV1NBubgCCKg --- ggml/src/ggml-cuda/fattn.cu | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu index 629a2e868117..c7cc0309e230 100644 --- a/ggml/src/ggml-cuda/fattn.cu +++ b/ggml/src/ggml-cuda/fattn.cu @@ -11,7 +11,7 @@ static void ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ggml_backend_cuda_con const ggml_tensor * Q = dst->src[0]; const ggml_tensor * V = dst->src[2]; - if (turing_mma_available(cc) && ggml_is_quantized(V->type) && Q->ne[1] > 2 && Q->ne[1] <= 4) { + if (type_KV == GGML_TYPE_F16 && turing_mma_available(cc) && ggml_is_quantized(V->type) && Q->ne[1] > 2 && Q->ne[1] <= 4) { ggml_cuda_flash_attn_ext_mma_f16_case(ctx, dst); return; }