Commit c2a9e1606 for llama.cpp
commit c2a9e1606807970f4ee3167bacd951699c89caea
Author: uvos <carl@uvos.xyz>
Date: Mon Sep 28 13:47:29 2026 +0200
HIP: fix template skip for DKQ > 256 mfma kernels (#29559)
diff --git a/ggml/src/ggml-cuda/fattn-mma-f16.cuh b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
index fa0d347af..083d3228a 100644
--- a/ggml/src/ggml-cuda/fattn-mma-f16.cuh
+++ b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
@@ -1875,7 +1875,7 @@ static __global__ void flash_attn_ext_f16(
#endif // defined(AMD_WMMA_AVAILABLE)
#if defined(AMD_MFMA_AVAILABLE)
- if (ncols1*ncols2 < 16 || (DKQ > 256 && ncols1*ncols2 < 64)) {
+ if (ncols1*ncols2 < 16 || (DKQ > 256 && ncols1*ncols2 < 32)) {
NO_DEVICE_CODE;
return;
}