Commit 000bee54a for llama.cpp
commit 000bee54a544070356b44d68af299b973d391d26
Author: cwriter <silvan.niederer@bluewin.ch>
Date: Thu Oct 8 08:30:00 2026 +0200
sycl: stage bulk uploads (model loading) through a pinned ring buffer (#29608)
Co-authored-by: cwriter <cwriter@localhost>
diff --git a/docs/backend/SYCL.md b/docs/backend/SYCL.md
index c2ae0c1ba..19dc69851 100644
--- a/docs/backend/SYCL.md
+++ b/docs/backend/SYCL.md
@@ -803,6 +803,7 @@ User can use the device management in [docs/multi-gpu.md](https://github.com/ggm
| GGML_SYCL_ENABLE_GRAPH | 0 (default) or 1 | Enable running computations through SYCL Graphs feature. Disabled by default because SYCL Graph is still on development, no better performance. |
| GGML_SYCL_ENABLE_HOST_PINNED_MEM | 0 or 1 (default) | Enable host pinned memory to speed up copy data from host to device. When disable it, host memory will common malloc() on CPU. Disable it when use `--load-model mlock`.|
| GGML_SYCL_HOST_PINNED_MEM_2G | 0 (default) or 1 | Limit the max memory allocation to be no more than 2GB when enable host pinned memory. USM allocations above 2 GiB take the relaxed/large-allocation path, which serializes H2D copies with compute and prevents copy/compute overlap. It will impact the startup time. Need more test. Depend on `GGML_SYCL_ENABLE_HOST_PINNED_MEM=1`.|
+| GGML_SYCL_UPLOAD_STAGING_SLOTS | 4 (default) or non-negative integer | Number of 8 MiB pinned host slots used to stage tensor uploads (model loading), so the host copy of one slot overlaps the transfer of the previous one. Set to 0 to use the old path: a malloc'd bounce buffer and a blocking copy per tensor. |
| GGML_SYCL_GET_MEM_API | 0 (default) or 1 | Set to get memory info (free, total) by Level Zero or SYCL API:<br>0 - Level Zero API: support more GPUs, only run on Level Zero running time. When there is an error, fallback to call SYCL API. Depend on GGML_SYCL_SUPPORT_LEVEL_ZERO_API.<br>1 - SYCL API: legacy, support more running time, it can't get the free size of some GPUs (like Arc770). In such case, return the free size as value of total size.|
| GGML_SYCL_USE_LEVEL_ZERO_API | 1 (default) or 0 | Use Level Zero API for device memory allocation instead of SYCL. Reduces system RAM usage on Intel dGPUs by avoiding DMA-buf/TTM host memory staging. Requires GGML_SYCL_SUPPORT_LEVEL_ZERO_API=ON at build time. SYCL backend always runs on Level Zero running time even if it's set as OFF (The SYCL api will be usage for memory allocation).|
| GGML_SYCL_ENABLE_DNN | 0 or 1 (default)| Enable running computations through oneDNN and always use oneMKL. |
diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp
index ebca92886..13fc02325 100644
--- a/ggml/src/ggml-sycl/ggml-sycl.cpp
+++ b/ggml/src/ggml-sycl/ggml-sycl.cpp
@@ -139,6 +139,7 @@ int g_ggml_sycl_dev2dev_memcpy = DEV2DEV_MEMCPY_SYCL;
int g_ggml_sycl_usm_system = 0;
int g_ggml_sycl_enable_host_pinned_mem = 1;
int g_ggml_sycl_host_pinned_mem_2g = 0;
+int g_ggml_sycl_upload_staging_slots = 4;
int g_ggml_sycl_get_mem_api = MEMORY_API_TYPE_LEVEL_ZERO;
int g_ggml_sycl_enable_sparse_fa = 0;
int g_ggml_sycl_debug_sparse_fa = 0;
@@ -458,6 +459,7 @@ static void ggml_check_sycl() try {
g_ggml_sycl_host_pinned_mem_2g =
ggml_sycl_get_env("GGML_SYCL_HOST_PINNED_MEM_2G", 0) & g_ggml_sycl_enable_host_pinned_mem;
+ g_ggml_sycl_upload_staging_slots = std::max(0, ggml_sycl_get_env("GGML_SYCL_UPLOAD_STAGING_SLOTS", 4));
g_ggml_sycl_enable_sparse_fa = ggml_sycl_get_env("GGML_SYCL_SPARSE_FA", 0);
g_ggml_sycl_debug_sparse_fa = ggml_sycl_get_env("GGML_SYCL_SPARSE_FA_DEBUG", 0);
@@ -555,6 +557,7 @@ static void ggml_check_sycl() try {
#endif
GGML_LOG_INFO(" GGML_SYCL_ENABLE_FUSION: %d\n", g_ggml_sycl_enable_fusion);
+ GGML_LOG_INFO(" GGML_SYCL_UPLOAD_STAGING_SLOTS: %d\n", g_ggml_sycl_upload_staging_slots);
#if defined(__INTEL_LLVM_COMPILER)
GGML_LOG_INFO(" GGML_SYCL_ENABLE_ESIMD: %d\n", g_ggml_sycl_enable_esimd);
@@ -667,12 +670,23 @@ inline void free_aligned_mem_host(void * memblock) {
// sycl buffer
struct ggml_backend_sycl_buffer_context {
+ // pinned staging for uploads; the host fills one slot while the previous one transfers
+ static constexpr size_t staging_slot_size = 8*1024*1024;
+
+ struct host_staging {
+ void * data = nullptr;
+ std::vector<sycl::event> events;
+ std::vector<bool> submitted;
+ int next = 0;
+ };
+
int device;
void * dev_ptr = nullptr;
queue_ptr stream;
std::string name;
optimize_feature opt_feature;
std::vector<ggml_tensor_extra_gpu *> tensor_extras;
+ host_staging staging;
bool is_usm_system;
ggml_backend_sycl_buffer_context(int device, void * dev_ptr, queue_ptr stream, bool is_usm_system) :
@@ -682,7 +696,22 @@ struct ggml_backend_sycl_buffer_context {
opt_feature = ggml_sycl_info().devices[device].opt_feature;
}
+ // waits for every queued upload, then releases the pinned block
+ void drop_host_staging() {
+ for (size_t i = 0; i < staging.submitted.size(); ++i) {
+ if (staging.submitted[i]) {
+ staging.events[i].wait_and_throw();
+ staging.submitted[i] = false;
+ }
+ }
+ if (staging.data != nullptr) {
+ sycl::free(staging.data, *stream);
+ staging.data = nullptr;
+ }
+ }
+
~ggml_backend_sycl_buffer_context() {
+ drop_host_staging();
if (dev_ptr != nullptr) {
ggml_sycl_set_device(device);
if (is_usm_system)
@@ -783,6 +812,40 @@ static void ggml_backend_sycl_buffer_set_tensor(ggml_backend_buffer_t buffer,
GGML_SYCL_DEBUG(" size=%zu offset=%zu\n", size, offset);
ggml_backend_sycl_buffer_context * ctx = ( ggml_backend_sycl_buffer_context *)buffer->context;
ggml_sycl_set_device(ctx->device);
+
+ // copy through pinned memory so the device never reads mmap()ed pages directly
+ // chunks pipeline on the in-order compute queue, so no drain per tensor is needed
+ const int n_slots = g_ggml_sycl_upload_staging_slots;
+ if (n_slots > 0 && ctx->staging.data == nullptr) {
+ ctx->staging.data = sycl::malloc_host(n_slots * ctx->staging_slot_size, *ctx->stream);
+ if (ctx->staging.data != nullptr) {
+ ctx->staging.events.resize(n_slots);
+ ctx->staging.submitted.assign(n_slots, false);
+ }
+ }
+ if (ctx->staging.data != nullptr) {
+ queue_ptr stream = ctx->stream;
+ char * dst = (char *) tensor->data + offset;
+ const char * src = (const char *) data;
+ size_t remaining = size;
+ while (remaining > 0) {
+ const size_t chunk = std::min(remaining, ctx->staging_slot_size);
+ const int slot = ctx->staging.next;
+ ctx->staging.next = (ctx->staging.next + 1) % (int) ctx->staging.submitted.size();
+ if (ctx->staging.submitted[slot]) {
+ ctx->staging.events[slot].wait_and_throw();
+ }
+ void * stage = (char *) ctx->staging.data + slot * ctx->staging_slot_size;
+ memcpy(stage, src, chunk);
+ ctx->staging.events[slot] = stream->memcpy(dst, stage, chunk);
+ ctx->staging.submitted[slot] = true;
+ src += chunk;
+ dst += chunk;
+ remaining -= chunk;
+ }
+ return;
+ }
+
auto stream = &(dpct::dev_mgr::instance().get_device(ctx->device).default_queue());
SYCL_CHECK(CHECK_TRY_ERROR(dpct::dev_mgr::instance().get_device(ctx->device).queues_wait_and_throw()));
#ifndef _WIN32