Commit 1bb2b9fcb for llama.cpp
commit 1bb2b9fcbe85c35c4e3b787e1c868ef05ca702f3
Author: uvos <carl@uvos.xyz>
Date: Sat Oct 10 08:14:51 2026 +0200
CUDA/HIP: fix race in flash_attn_ext_f16_process_tile when nbatch_combine != DKQ/2 (#30103)
Suggested-by: Johannes Gäßler <johannesg@5d6.de>
diff --git a/ggml/src/ggml-cuda/fattn-mma-f16.cuh b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
index 7f0f6abfd..296aa2b34 100644
--- a/ggml/src/ggml-cuda/fattn-mma-f16.cuh
+++ b/ggml/src/ggml-cuda/fattn-mma-f16.cuh
@@ -1772,7 +1772,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile(
}
}
}
- if (np > 1) {
+ if (np > 1 || nbatch_combine != DKQ/2) {
__syncthreads();
}
}