From 19836568c4f1b504d8ee579b7ab7fcf4edbd8f78 Mon Sep 17 00:00:00 2001 From: Gabriel Date: Fri, 18 Sep 2026 20:33:25 +0300 Subject: [PATCH] cuda: avoid short-query MMA FA crash with quantized V --- ggml/src/ggml-cuda/fattn.cu | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu index ab7a3b297c07..b6ef1f141c29 100644 --- a/ggml/src/ggml-cuda/fattn.cu +++ b/ggml/src/ggml-cuda/fattn.cu @@ -9,6 +9,12 @@ template static void ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; const ggml_tensor * Q = dst->src[0]; + const ggml_tensor * V = dst->src[2]; + + if (turing_mma_available(cc) && ggml_is_quantized(V->type) && Q->ne[1] > 2 && Q->ne[1] <= 4) { + ggml_cuda_flash_attn_ext_mma_f16_case(ctx, dst); + return; + } if constexpr (ncols2 <= 8) { if (turing_mma_available(cc) && Q->ne[1] <= 8/ncols2) {