Commit 005a1e127 for llama.cpp
commit 005a1e127a84cc75bb66be11b43e4f0773564e53
Author: Neo Zhang <zhang.jianyu@outlook.com>
Date: Wed Oct 7 15:24:43 2026 +0800
[SYCL] fix the issue in mixed different model GPUs in FA (#29071)
* fix mixed different model GPUs issue
* Update ggml/src/ggml-sycl/ggml-sycl.cpp
Co-authored-by: Georgi Gerganov <ggerganov@gmail.com>
---------
Co-authored-by: Georgi Gerganov <ggerganov@gmail.com>
diff --git a/ggml/src/ggml-sycl/fattn-vec.hpp b/ggml/src/ggml-sycl/fattn-vec.hpp
index 9ec88c287..5e40c88fb 100644
--- a/ggml/src/ggml-sycl/fattn-vec.hpp
+++ b/ggml/src/ggml-sycl/fattn-vec.hpp
@@ -589,10 +589,10 @@ void ggml_sycl_flash_attn_ext_vec_case_impl(ggml_backend_sycl_context & ctx, ggm
const bool need_f16_V = type_V == GGML_TYPE_F16;
constexpr size_t nbytes_shared = 0;
+ const auto arch = ggml_sycl_info().devices[ggml_sycl_get_device()].hw_info.arch;
// D=512 does not fit the default register file; it spills up to 343 bytes per thread, against at most 57 for D <= 256. This kernel is decode only, so thread occupancy is not the limit. It is 1.9x faster at every KV depth on Battlemage.
constexpr bool use_large_grf = D >= 512;
- const auto arch = ggml_sycl_info().devices[ctx.device].hw_info.arch;
const int nthreads = ggml_sycl_fattn_vec_get_nthreads_device(arch);
if constexpr (D <= 256) {
if (nthreads == 256) {
diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp
index c6aff3955..b4d200526 100644
--- a/ggml/src/ggml-sycl/ggml-sycl.cpp
+++ b/ggml/src/ggml-sycl/ggml-sycl.cpp
@@ -792,6 +792,14 @@ static bool ggml_sycl_is_l0_discrete_gpu(int device) {
}
#endif
+static void memcpy_host_forward(sycl::queue &q_dst, sycl::queue &q_src, void *ptr_dst,
+ const void *ptr_src, size_t size) {
+ char *host_buf = (char *)malloc(size);
+ q_src.memcpy(host_buf, (const char *)ptr_src, size).wait();
+ q_dst.memcpy((char *)ptr_dst, host_buf, size).wait();
+ free(host_buf);
+}
+
static void dev2dev_memcpy(int device_dst, sycl::queue &q_dst, int device_src, sycl::queue &q_src, void *ptr_dst,
const void *ptr_src, size_t size) {
@@ -835,10 +843,7 @@ static void dev2dev_memcpy(int device_dst, sycl::queue &q_dst, int device_src, s
} else {
GGML_SYCL_DEBUG("[SYCL] dev2dev memcpy by host forward for SYCL/L0 fallback\n");
}
- char *host_buf = (char *)malloc(size);
- q_src.memcpy(host_buf, (const char *)ptr_src, size).wait();
- q_dst.memcpy((char *)ptr_dst, host_buf, size).wait();
- free(host_buf);
+ memcpy_host_forward(q_dst, q_src, ptr_dst, ptr_src, size);
}
static bool