Commit c811cb8f0 for llama.cpp

commit c811cb8f0ac91b8ac72a32f970bdd45037f20da7
Author: Aman Gupta <amangupta052@gmail.com>
Date:   Thu Oct 8 15:50:49 2026 +0530

    llama: support MoE cache over multiple GPUs (#30112)

diff --git a/common/arg.cpp b/common/arg.cpp
index 182f76f45..9efcf79ac 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -2778,7 +2778,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
     ).set_env("LLAMA_ARG_N_CPU_MOE"));
     add_opt(common_arg(
         {"--moe-cache-mib"}, "N",
-        "GPU cache size in MiB for the MoE experts kept in the CPU (default: 0, disabled)",
+        "GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)",
         [](common_params & params, int value) {
             if (value < 0) {
                 throw std::invalid_argument("invalid value");
diff --git a/common/common.h b/common/common.h
index de88dfb9b..ef062bb84 100644
--- a/common/common.h
+++ b/common/common.h
@@ -593,7 +593,7 @@ struct common_params {
     ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K
     ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V

-    size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU
+    size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU, split among the GPUs like the layers

     common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;

diff --git a/include/llama.h b/include/llama.h
index 77f527d5b..60329024f 100644
--- a/include/llama.h
+++ b/include/llama.h
@@ -396,7 +396,7 @@ extern "C" {
         enum ggml_type type_k; // data type for K cache [EXPERIMENTAL]
         enum ggml_type type_v; // data type for V cache [EXPERIMENTAL]

-        size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, 0 = disabled [EXPERIMENTAL]
+        size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, split among the devices like the layers, 0 = disabled [EXPERIMENTAL]

         // Abort callback
         // if it returns true, execution of llama_decode() will be aborted
diff --git a/src/llama-context.cpp b/src/llama-context.cpp
index 5f861f96c..9dcaf05a6 100644
--- a/src/llama-context.cpp
+++ b/src/llama-context.cpp
@@ -436,7 +436,8 @@ llama_context::llama_context(
             model.n_gpu_layers() > model.hparams.n_layer_all &&
             model.split_mode() == LLAMA_SPLIT_MODE_LAYER &&
             cparams.offload_kqv &&
-            !model.has_tensor_overrides();
+            !model.has_tensor_overrides() &&
+            cparams.moe_cache_size == 0; // not supported by the MoE cache

         // pipeline parallelism requires support for async compute and events in all devices
         if (pipeline_parallel) {
@@ -465,19 +466,7 @@ llama_context::llama_context(
         }

         if (cparams.moe_cache_size > 0) {
-            if (cparams.pipeline_parallel || model.n_devices() > 1) {
-                throw std::runtime_error("MoE cache does not support multiple devices");
-            }
-            for (size_t i = 0; i < backend_ptrs.size(); ++i) {
-                const auto type = ggml_backend_dev_type(ggml_backend_get_device(backend_ptrs[i]));
-                if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
-                    moe_cache = std::make_unique<llama_moe_cache>(model, backend_ptrs[i], backend_buft[i], cparams.moe_cache_size);
-                    break;
-                }
-            }
-            if (!moe_cache) {
-                throw std::runtime_error("MoE cache requires a GPU backend");
-            }
+            moe_cache = std::make_unique<llama_moe_cache>(model, backend_ptrs, backend_buft, cparams.moe_cache_size);
         }

         sched_reserve();
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index f030eb9e6..23e4c9091 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -2448,10 +2448,10 @@ ggml_tensor * llm_graph_context::build_moe_cache_slots(
     // the slot map is a host weight, so the scheduler starts a new split here and copies it with the copy callback
     // the callback reads the selected experts, uploads the missing ones and updates the slot map
     ggml_tensor * slots = ggml_get_rows(ctx0, slot_map, ids); // [1, n_expert_used*n_tokens]
-    if (!ggml_backend_supports_op(moe_cache->backend(), slots)) {
+    if (!ggml_backend_supports_op(moe_cache->backend(il), slots)) {
         return nullptr;
     }
-    ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend());
+    ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend(il));
     cb(slots, "ffn_moe_slots", il);

     return ggml_reshape_2d(ctx0, slots, selected_experts->ne[0], selected_experts->ne[1]); // [n_expert_used, n_tokens]
diff --git a/src/llama-moe-cache.cpp b/src/llama-moe-cache.cpp
index 7292a1e57..f22d986dc 100644
--- a/src/llama-moe-cache.cpp
+++ b/src/llama-moe-cache.cpp
@@ -151,8 +151,22 @@ static bool llama_moe_cache_is_host_weight(const ggml_tensor * t) {
 }

 struct llama_moe_cache::impl {
-    // layers with the same expert tensor layout share the banks and the LRU of a group
+    // a GPU with its own budget and banks, it caches the layers assigned to it
+    struct device {
+        ggml_backend_t backend;
+        ggml_backend_buffer_type_t buft;
+        size_t host_bytes = 0; // host experts of the layers it caches
+        double split = 0.0;    // share of the budget
+
+        // banks and their views
+        ggml_context_ptr ctx;
+        ggml_backend_buffer_ptr buf;
+        size_t buf_size = 0;
+    };
+
+    // layers of the same device with the same expert tensor layout share the banks and the LRU of a group
     struct group {
+        int32_t id;                       // device
         std::vector<ggml_tensor *> ref;   // expert tensors of the first layer
         std::vector<int32_t> layers;
         std::vector<ggml_tensor *> banks; // device storage of all slots, one per expert tensor
@@ -181,13 +195,13 @@ struct llama_moe_cache::impl {

     static constexpr int64_t max_batch = 32;

-    ggml_backend_t backend;
     int32_t n_expert_used;

     stats stats_small; // up to 8 tokens per ubatch
     stats stats_large;
     stats stats_copy;  // experts copied from the cache for large batches

+    std::vector<device> devices;
     std::vector<group> groups;
     std::vector<layer> layers;
     std::unordered_map<const ggml_tensor *, binding> bindings; // host experts -> cached experts
@@ -196,11 +210,6 @@ struct llama_moe_cache::impl {
     std::vector<int32_t> ids;
     std::vector<moe_cache_lru::fill> fills;

-    // banks and their views on the device
-    ggml_context_ptr ctx;
-    ggml_backend_buffer_ptr buf;
-    size_t buf_size = 0;
-
     // slot maps in host memory
     ggml_context_ptr ctx_host;
     ggml_backend_buffer_ptr buf_host;
@@ -209,11 +218,17 @@ struct llama_moe_cache::impl {
     // views used by copy_experts
     ggml_context_ptr ctx_views;

-    impl(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
-            backend(backend), n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) {
-        ggml_backend_dev_t dev = ggml_backend_get_device(backend);
-        const auto dev_type = ggml_backend_dev_type(dev);
-        if (dev_type != GGML_BACKEND_DEVICE_TYPE_GPU && dev_type != GGML_BACKEND_DEVICE_TYPE_IGPU) {
+    impl(const llama_model & model, const std::vector<ggml_backend_t> & backends, const std::vector<ggml_backend_buffer_type_t> & bufts, size_t size) :
+            n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) {
+        for (size_t i = 0; i < backends.size(); ++i) {
+            const auto dev_type = ggml_backend_dev_type(ggml_backend_get_device(backends[i]));
+            if (dev_type == GGML_BACKEND_DEVICE_TYPE_GPU || dev_type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
+                auto & d = devices.emplace_back();
+                d.backend = backends[i];
+                d.buft    = bufts[i];
+            }
+        }
+        if (devices.empty()) {
             throw std::runtime_error("MoE cache requires a GPU backend");
         }
         if (model.split_mode() == LLAMA_SPLIT_MODE_TENSOR) {
@@ -223,24 +238,28 @@ struct llama_moe_cache::impl {
             throw std::runtime_error("MoE cache requires a MoE model");
         }

-        // only cache layers that keep all of their experts in host memory
-        size_t host_bytes = 0;
+        // only cache layers that keep all of their experts in host memory, on the device the layer is assigned to
         for (size_t il = 0; il < model.layers.size(); ++il) {
             auto experts = llama_moe_cache_layer_experts(model.layers[il]);
-            if (experts.empty() || model.dev_layer(il) != dev ||
-                !std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) {
+            if (experts.empty() || !std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) {
+                continue;
+            }
+            const auto it_dev = std::find_if(devices.begin(), devices.end(), [&](const device & d) { return ggml_backend_get_device(d.backend) == model.dev_layer(il); });
+            if (it_dev == devices.end()) {
                 continue;
             }
-            auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return llama_moe_cache_same_layout(g.ref, experts); });
+            const int32_t id = (int32_t) (it_dev - devices.begin());
+            auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return g.id == id && llama_moe_cache_same_layout(g.ref, experts); });
             if (it == groups.end()) {
                 groups.emplace_back();
                 it = groups.end() - 1;
+                it->id  = id;
                 it->ref = experts;
             }
             it->layers.push_back(il);
             for (const ggml_tensor * t : experts) {
-                it->host_bytes += ggml_nbytes(t);
-                host_bytes     += ggml_nbytes(t);
+                it->host_bytes         += ggml_nbytes(t);
+                devices[id].host_bytes += ggml_nbytes(t);
             }
         }
         if (groups.empty()) {
@@ -249,8 +268,8 @@ struct llama_moe_cache::impl {
         }

         // one extra slot at the end, CUDA MMQ can read past the last expert
-        const size_t alignment = ggml_backend_buft_get_alignment(buft);
         auto alloc_size = [&](const group & g, int32_t n_slots) {
+            const size_t alignment = ggml_backend_buft_get_alignment(devices[g.id].buft);
             size_t res = 0;
             for (const ggml_tensor * t : g.ref) {
                 res += GGML_PAD(t->nb[2]*(n_slots + 1), alignment);
@@ -258,12 +277,43 @@ struct llama_moe_cache::impl {
             return res;
         };

-        // split the budget by the size of the experts, so each group caches the same fraction of its experts
-        size_t n_tensors      = 0;
+        // the budget is split among the devices with host experts like the layers, by the tensor split or by default by free memory
+        const float * tensor_split = model.tensor_split();
+        const bool split_by_free = tensor_split == nullptr ||
+            std::all_of(tensor_split, tensor_split + model.n_devices(), [](float x) { return x == 0.0f; });
+        double split_sum = 0.0;
+        for (device & d : devices) {
+            if (d.host_bytes == 0) {
+                continue;
+            }
+            ggml_backend_dev_t dev = ggml_backend_get_device(d.backend);
+            if (split_by_free) {
+                size_t free;
+                size_t total;
+                ggml_backend_dev_memory(dev, &free, &total);
+                d.split = (double) free;
+            } else {
+                const auto it = std::find_if(model.devices.begin(), model.devices.end(), [&](const llama_device & ld) { return ld.dev == dev; });
+                GGML_ASSERT(it != model.devices.end());
+                d.split = (double) tensor_split[it - model.devices.begin()];
+            }
+            split_sum += d.split;
+        }
+        if (split_sum == 0.0) {
+            // the devices do not report their free memory
+            for (device & d : devices) {
+                d.split    = d.host_bytes > 0 ? 1.0 : 0.0;
+                split_sum += d.split;
+            }
+        }
+
+        // within a device the budget is split by the size of the experts, so each group caches the same fraction of its experts
+        std::vector<size_t> n_tensors(devices.size(), 0);
         size_t n_tensors_host = 0;
         for (group & g : groups) {
+            const device & d = devices[g.id];
             const int32_t n_expert  = g.ref[0]->ne[2];
-            const size_t  budget    = (size_t) ((double) size*g.host_bytes/host_bytes);
+            const size_t  budget    = (size_t) ((double) size*d.split/split_sum*g.host_bytes/d.host_bytes);
             const int32_t max_slots = g.layers.size()*n_expert;
             while (g.n_slots < max_slots && alloc_size(g, g.n_slots + 1) <= budget) {
                 g.n_slots++;
@@ -274,10 +324,10 @@ struct llama_moe_cache::impl {
                 continue;
             }
             g.lru.init(model.layers.size(), n_expert, g.n_slots);
-            n_tensors      += g.ref.size()*(1 + g.layers.size());
-            n_tensors_host += g.layers.size();
+            n_tensors[g.id] += g.ref.size()*(1 + g.layers.size());
+            n_tensors_host  += g.layers.size();
         }
-        if (n_tensors == 0) {
+        if (n_tensors_host == 0) {
             throw std::runtime_error("MoE cache is too small to hold the experts of one token");
         }

@@ -293,7 +343,11 @@ struct llama_moe_cache::impl {
             }
             return res;
         };
-        ctx       = init_ctx(n_tensors);
+        for (size_t id = 0; id < devices.size(); ++id) {
+            if (n_tensors[id] > 0) {
+                devices[id].ctx = init_ctx(n_tensors[id]);
+            }
+        }
         ctx_host  = init_ctx(n_tensors_host);
         ctx_views = init_ctx(2);

@@ -305,8 +359,9 @@ struct llama_moe_cache::impl {
             if (g.n_slots == 0) {
                 continue;
             }
+            ggml_context * ctx = devices[g.id].ctx.get();
             for (const ggml_tensor * t : g.ref) {
-                ggml_tensor * bank = ggml_new_tensor_3d(ctx.get(), t->type, t->ne[0], t->ne[1], g.n_slots + 1);
+                ggml_tensor * bank = ggml_new_tensor_3d(ctx, t->type, t->ne[0], t->ne[1], g.n_slots + 1);
                 GGML_ASSERT(bank->nb[2] == t->nb[2]);
                 ggml_format_name(bank, "moe_cache.%zu.%s", ig, t->name);
                 g.banks.push_back(bank);
@@ -317,7 +372,7 @@ struct llama_moe_cache::impl {
                 l.experts = llama_moe_cache_layer_experts(model.layers[il]);
                 for (size_t ip = 0; ip < l.experts.size(); ++ip) {
                     ggml_tensor * bank   = g.banks[ip];
-                    ggml_tensor * cached = ggml_view_3d(ctx.get(), bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0);
+                    ggml_tensor * cached = ggml_view_3d(ctx, bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0);
                     ggml_format_name(cached, "moe_cache.%s", l.experts[ip]->name);
                     bindings[l.experts[ip]] = { il, (int32_t) ip, cached };
                 }
@@ -326,28 +381,41 @@ struct llama_moe_cache::impl {
                 layer_of[l.slot_map] = il;
                 buf_host_size += GGML_PAD(ggml_nbytes(l.slot_map), alignment_host);
             }
-            buf_size += alloc_size(g, g.n_slots);
+            devices[g.id].buf_size += alloc_size(g, g.n_slots);
         }

         if (model.hparams.no_alloc) {
             // only used to measure the memory use, see llama_context::memory_breakdown
-            buf.reset(ggml_backend_buft_alloc_buffer(buft, 0));
-            buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0));
-            for (ggml_tensor * t = ggml_get_first_tensor(ctx.get()); t != nullptr; t = ggml_get_next_tensor(ctx.get(), t)) {
-                t->buffer = buf.get();
+            for (device & d : devices) {
+                if (!d.ctx) {
+                    continue;
+                }
+                d.buf.reset(ggml_backend_buft_alloc_buffer(d.buft, 0));
+                for (ggml_tensor * t = ggml_get_first_tensor(d.ctx.get()); t != nullptr; t = ggml_get_next_tensor(d.ctx.get(), t)) {
+                    t->buffer = d.buf.get();
+                }
             }
+            buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0));
             for (ggml_tensor * t = ggml_get_first_tensor(ctx_host.get()); t != nullptr; t = ggml_get_next_tensor(ctx_host.get(), t)) {
                 t->buffer = buf_host.get();
             }
         } else {
-            buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx.get(), buft));
+            for (device & d : devices) {
+                if (!d.ctx) {
+                    continue;
+                }
+                d.buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(d.ctx.get(), d.buft));
+                if (!d.buf) {
+                    throw std::runtime_error("failed to allocate the MoE cache buffers");
+                }
+                ggml_backend_buffer_clear(d.buf.get(), 0);
+                d.buf_size = ggml_backend_buffer_get_size(d.buf.get());
+            }
             buf_host.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx_host.get(), buft_host));
-            if (!buf || !buf_host) {
+            if (!buf_host) {
                 throw std::runtime_error("failed to allocate the MoE cache buffers");
             }
-            ggml_backend_buffer_clear(buf.get(), 0);
             ggml_backend_buffer_clear(buf_host.get(), 0xff); // all slots are -1
-            buf_size      = ggml_backend_buffer_get_size(buf.get());
             buf_host_size = ggml_backend_buffer_get_size(buf_host.get());

             for (group & g : groups) {
@@ -360,17 +428,33 @@ struct llama_moe_cache::impl {
         }

         // as weights, the ops that read the banks run on the device and the slot maps are copied with the copy callback
-        ggml_backend_buffer_set_usage(buf.get(),      GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
+        for (device & d : devices) {
+            if (d.buf) {
+                ggml_backend_buffer_set_usage(d.buf.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
+            }
+        }
         ggml_backend_buffer_set_usage(buf_host.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);

-        LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__,
-            ggml_backend_buft_name(buft), buf_size/1024.0/1024.0, host_bytes/1024.0/1024.0);
-        for (const group & g : groups) {
-            LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__,
-                g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2]));
+        for (size_t id = 0; id < devices.size(); ++id) {
+            const device & d = devices[id];
+            if (d.host_bytes == 0) {
+                continue;
+            }
+            LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__,
+                ggml_backend_buft_name(d.buft), d.buf_size/1024.0/1024.0, d.host_bytes/1024.0/1024.0);
+            for (const group & g : groups) {
+                if (g.id == (int32_t) id) {
+                    LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__,
+                        g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2]));
+                }
+            }
         }
     }

+    ggml_backend_t backend(int32_t il) const {
+        return devices[groups[layers[il].ig].id].backend;
+    }
+
     ~impl() {
         log_stats();
     }
@@ -395,11 +479,14 @@ struct llama_moe_cache::impl {

     int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
         const auto it = bindings.find(w);
-        if (it == bindings.end() || backend != this->backend) {
+        if (it == bindings.end()) {
             return 0;
         }
         const binding & b = it->second;
         const group & g = groups[layers[b.il].ig];
+        if (backend != devices[g.id].backend) {
+            return 0;
+        }

         // large batches only read the cache, so the experts used in generation stay in it
         const int32_t * slots = g.lru.slot_map[b.il];
@@ -434,7 +521,7 @@ struct llama_moe_cache::impl {
         const layer & l = layers[il];
         group & g = groups[l.ig];

-        GGML_ASSERT(backend == this->backend);
+        GGML_ASSERT(backend == devices[g.id].backend);

         // the get_rows that looks up the slots of the selected experts
         const int n_nodes = ggml_graph_n_nodes(graph);
@@ -512,14 +599,14 @@ struct llama_moe_cache::impl {
     }
 };

-llama_moe_cache::llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
-    pimpl(new impl(model, backend, buft, size)) {
+llama_moe_cache::llama_moe_cache(const llama_model & model, const std::vector<ggml_backend_t> & backends, const std::vector<ggml_backend_buffer_type_t> & bufts, size_t size) :
+    pimpl(new impl(model, backends, bufts, size)) {
 }

 llama_moe_cache::~llama_moe_cache() = default;

-ggml_backend_t llama_moe_cache::backend() const {
-    return pimpl->backend;
+ggml_backend_t llama_moe_cache::backend(int32_t il) const {
+    return pimpl->backend(il);
 }

 ggml_tensor * llama_moe_cache::get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
@@ -540,8 +627,10 @@ int64_t llama_moe_cache::copy_experts(ggml_backend_t backend, const ggml_tensor

 std::map<ggml_backend_buffer_type_t, size_t> llama_moe_cache::memory_breakdown() const {
     std::map<ggml_backend_buffer_type_t, size_t> res;
-    if (pimpl->buf) {
-        res[ggml_backend_buffer_get_type(pimpl->buf.get())] += pimpl->buf_size;
+    for (const auto & d : pimpl->devices) {
+        if (d.buf) {
+            res[ggml_backend_buffer_get_type(d.buf.get())] += d.buf_size;
+        }
     }
     if (pimpl->buf_host) {
         res[ggml_backend_buffer_get_type(pimpl->buf_host.get())] += pimpl->buf_host_size;
diff --git a/src/llama-moe-cache.h b/src/llama-moe-cache.h
index 4872370d4..3b6726b86 100644
--- a/src/llama-moe-cache.h
+++ b/src/llama-moe-cache.h
@@ -4,6 +4,7 @@

 #include <map>
 #include <memory>
+#include <vector>

 struct llama_model;

@@ -11,10 +12,12 @@ struct llama_model;
 // each layer has a slot map in host memory: when the scheduler copies it to the device, the copy callback uploads the missing experts
 class llama_moe_cache {
 public:
-    llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size);
+    // backends are all the backends of the context, each GPU gets its own cache of the given size for the layers assigned to it
+    llama_moe_cache(const llama_model & model, const std::vector<ggml_backend_t> & backends, const std::vector<ggml_backend_buffer_type_t> & bufts, size_t size);
     ~llama_moe_cache();

-    ggml_backend_t backend() const;
+    // the device that caches layer il
+    ggml_backend_t backend(int32_t il) const;

     // the slot map of layer il, if its experts can be read from the cache for n_tokens tokens, nullptr otherwise
     ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const;
diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp
index b54b3cc18..78128ebf0 100644
--- a/tests/test-llama-archs.cpp
+++ b/tests/test-llama-archs.cpp
@@ -919,6 +919,11 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
             if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
                 dev_configs.emplace_back(std::vector<ggml_backend_dev_t>{devices_meta[0]}, "MoE cache", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024);
                 max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
+                // each GPU caches the layers assigned to it
+                if (devices_meta.size() > 1) {
+                    dev_configs.emplace_back(devices_meta, "MoE cache, layer split", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024);
+                    max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
+                }
             }
         }
     }
diff --git a/tools/cli/README.md b/tools/cli/README.md
index acecc74a6..20cf01c6b 100644
--- a/tools/cli/README.md
+++ b/tools/cli/README.md
@@ -63,6 +63,7 @@
 | `-ot, --override-tensor <tensor name pattern>=<buffer type>,...` | override tensor buffer type<br/>(env: LLAMA_ARG_OVERRIDE_TENSOR) |
 | `-cmoe, --cpu-moe` | keep all Mixture of Experts (MoE) weights in the CPU<br/>(env: LLAMA_ARG_CPU_MOE) |
 | `-ncmoe, --n-cpu-moe N` | keep the Mixture of Experts (MoE) weights of the first N layers in the CPU<br/>(env: LLAMA_ARG_N_CPU_MOE) |
+| `--moe-cache-mib N` | GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)<br/>(env: LLAMA_ARG_MOE_CACHE_MIB) |
 | `-ncffn, --n-cpu-ffn N` | keep the dense FFN weights of the first N layers in the CPU<br/>(dense models; for MoE expert weights use --n-cpu-moe)<br/>(env: LLAMA_ARG_N_CPU_FFN) |
 | `-ngl, --gpu-layers, --n-gpu-layers N` | max. number of layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS) |
 | `-sm, --split-mode {none,layer,row,tensor}` | how to split the model across multiple GPUs, one of:<br/>- none: use one GPU only<br/>- layer (default): split layers and KV across GPUs (pipelined)<br/>- row: split weight across GPUs by rows (parallelized)<br/>- tensor: split weights and KV across GPUs (parallelized, EXPERIMENTAL)<br/>(env: LLAMA_ARG_SPLIT_MODE) |
@@ -204,6 +205,7 @@
 | `--spec-draft-p-split, --draft-p-split P` | speculative decoding split probability (default: 0.10)<br/>(env: LLAMA_ARG_SPEC_DRAFT_P_SPLIT) |
 | `--spec-draft-p-min, --draft-p-min P` | minimum speculative decoding probability (greedy) (default: 0.00)<br/>(env: LLAMA_ARG_SPEC_DRAFT_P_MIN) |
 | `--spec-draft-backend-sampling, --no-spec-draft-backend-sampling` | offload draft sampling to the backend (default: enabled)<br/>(env: LLAMA_ARG_SPEC_DRAFT_BACKEND_SAMPLING) |
+| `--spec-draft-sampling {greedy,probabilistic}` | how the draft is sampled: greedy takes its argmax, probabilistic samples it and has the target verify by rejection sampling (default: greedy)<br/>(env: LLAMA_ARG_SPEC_DRAFT_SAMPLING) |
 | `--spec-draft-device, -devd, --device-draft <dev1,dev2,..>` | comma-separated list of devices to use for offloading the draft model (none = don't offload, default: follows --device)<br/>use --list-devices to see a list of available devices |
 | `--spec-draft-ngl, -ngld, --gpu-layers-draft, --n-gpu-layers-draft N` | max. number of draft model layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS_DRAFT) |
 | `--spec-draft-model, -md, --model-draft FNAME` | draft model for speculative decoding (default: unused)<br/>(env: LLAMA_ARG_SPEC_DRAFT_MODEL) |
diff --git a/tools/completion/README.md b/tools/completion/README.md
index e2ac0668c..c396609ab 100644
--- a/tools/completion/README.md
+++ b/tools/completion/README.md
@@ -146,6 +146,7 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
 | `-ot, --override-tensor <tensor name pattern>=<buffer type>,...` | override tensor buffer type<br/>(env: LLAMA_ARG_OVERRIDE_TENSOR) |
 | `-cmoe, --cpu-moe` | keep all Mixture of Experts (MoE) weights in the CPU<br/>(env: LLAMA_ARG_CPU_MOE) |
 | `-ncmoe, --n-cpu-moe N` | keep the Mixture of Experts (MoE) weights of the first N layers in the CPU<br/>(env: LLAMA_ARG_N_CPU_MOE) |
+| `--moe-cache-mib N` | GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)<br/>(env: LLAMA_ARG_MOE_CACHE_MIB) |
 | `-ncffn, --n-cpu-ffn N` | keep the dense FFN weights of the first N layers in the CPU<br/>(dense models; for MoE expert weights use --n-cpu-moe)<br/>(env: LLAMA_ARG_N_CPU_FFN) |
 | `-ngl, --gpu-layers, --n-gpu-layers N` | max. number of layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS) |
 | `-sm, --split-mode {none,layer,row,tensor}` | how to split the model across multiple GPUs, one of:<br/>- none: use one GPU only<br/>- layer (default): split layers and KV across GPUs (pipelined)<br/>- row: split weight across GPUs by rows (parallelized)<br/>- tensor: split weights and KV across GPUs (parallelized, EXPERIMENTAL)<br/>(env: LLAMA_ARG_SPLIT_MODE) |
diff --git a/tools/server/README.md b/tools/server/README.md
index 9231f06e8..08d6f17a4 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -80,6 +80,7 @@ For the full list of features, please refer to [server's changelog](https://gith
 | `-ot, --override-tensor <tensor name pattern>=<buffer type>,...` | override tensor buffer type<br/>(env: LLAMA_ARG_OVERRIDE_TENSOR) |
 | `-cmoe, --cpu-moe` | keep all Mixture of Experts (MoE) weights in the CPU<br/>(env: LLAMA_ARG_CPU_MOE) |
 | `-ncmoe, --n-cpu-moe N` | keep the Mixture of Experts (MoE) weights of the first N layers in the CPU<br/>(env: LLAMA_ARG_N_CPU_MOE) |
+| `--moe-cache-mib N` | GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)<br/>(env: LLAMA_ARG_MOE_CACHE_MIB) |
 | `-ncffn, --n-cpu-ffn N` | keep the dense FFN weights of the first N layers in the CPU<br/>(dense models; for MoE expert weights use --n-cpu-moe)<br/>(env: LLAMA_ARG_N_CPU_FFN) |
 | `-ngl, --gpu-layers, --n-gpu-layers N` | max. number of layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS) |
 | `-sm, --split-mode {none,layer,row,tensor}` | how to split the model across multiple GPUs, one of:<br/>- none: use one GPU only<br/>- layer (default): split layers and KV across GPUs (pipelined)<br/>- row: split weight across GPUs by rows (parallelized)<br/>- tensor: split weights and KV across GPUs (parallelized, EXPERIMENTAL)<br/>(env: LLAMA_ARG_SPLIT_MODE) |
@@ -265,6 +266,7 @@ For the full list of features, please refer to [server's changelog](https://gith
 | `--spec-draft-p-split, --draft-p-split P` | speculative decoding split probability (default: 0.10)<br/>(env: LLAMA_ARG_SPEC_DRAFT_P_SPLIT) |
 | `--spec-draft-p-min, --draft-p-min P` | minimum speculative decoding probability (greedy) (default: 0.00)<br/>(env: LLAMA_ARG_SPEC_DRAFT_P_MIN) |
 | `--spec-draft-backend-sampling, --no-spec-draft-backend-sampling` | offload draft sampling to the backend (default: enabled)<br/>(env: LLAMA_ARG_SPEC_DRAFT_BACKEND_SAMPLING) |
+| `--spec-draft-sampling {greedy,probabilistic}` | how the draft is sampled: greedy takes its argmax, probabilistic samples it and has the target verify by rejection sampling (default: greedy)<br/>(env: LLAMA_ARG_SPEC_DRAFT_SAMPLING) |
 | `--spec-draft-device, -devd, --device-draft <dev1,dev2,..>` | comma-separated list of devices to use for offloading the draft model (none = don't offload, default: follows --device)<br/>use --list-devices to see a list of available devices |
 | `--spec-draft-ngl, -ngld, --gpu-layers-draft, --n-gpu-layers-draft N` | max. number of draft model layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS_DRAFT) |
 | `--spec-draft-model, -md, --model-draft FNAME` | draft model for speculative decoding (default: unused)<br/>(env: LLAMA_ARG_SPEC_DRAFT_MODEL) |