Commit c2a9e1606 for llama.cpp

commit c2a9e1606807970f4ee3167bacd951699c89caea
Author: uvos <carl@uvos.xyz>
Date:   Mon Sep 28 13:47:29 2026 +0200

    HIP: fix template skip for DKQ > 256 mfma kernels (#29559)

diff --git a/ggml/src/ggml-cuda/fattn-mma-f16.cuh b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
index fa0d347af..083d3228a 100644
--- a/ggml/src/ggml-cuda/fattn-mma-f16.cuh
+++ b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
@@ -1875,7 +1875,7 @@ static __global__ void flash_attn_ext_f16(
 #endif // defined(AMD_WMMA_AVAILABLE)

 #if defined(AMD_MFMA_AVAILABLE)
-    if (ncols1*ncols2 < 16 || (DKQ > 256 && ncols1*ncols2 < 64)) {
+    if (ncols1*ncols2 < 16 || (DKQ > 256 && ncols1*ncols2 < 32)) {
         NO_DEVICE_CODE;
         return;
     }