Commit 1bb2b9fcb for llama.cpp

commit 1bb2b9fcbe85c35c4e3b787e1c868ef05ca702f3
Author: uvos <carl@uvos.xyz>
Date:   Sat Oct 10 08:14:51 2026 +0200

    CUDA/HIP: fix race in flash_attn_ext_f16_process_tile when nbatch_combine != DKQ/2 (#30103)

    Suggested-by: Johannes Gäßler <johannesg@5d6.de>

diff --git a/ggml/src/ggml-cuda/fattn-mma-f16.cuh b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
index 7f0f6abfd..296aa2b34 100644
--- a/ggml/src/ggml-cuda/fattn-mma-f16.cuh
+++ b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
@@ -1772,7 +1772,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile(
                 }
             }
         }
-        if (np > 1) {
+        if (np > 1 || nbatch_combine != DKQ/2) {
             __syncthreads();
         }
     }