Commit d6cf9acb2 for llama.cpp

commit d6cf9acb25e6c657d6c6a4645ab7b69fce119995
Author: Aman Gupta <amangupta052@gmail.com>
Date:   Wed Oct 7 23:37:45 2026 +0530

    llama : add a GPU cache for MoE experts kept in host memory (#29887)

    * llama : add a GPU cache for MoE experts kept in host memory

    Assisted-by: Claude

    * use llama_moe_cache_ptr

diff --git a/common/arg.cpp b/common/arg.cpp
index 14d82f695..182f76f45 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -2776,6 +2776,16 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
             llm_add_n_cpu_ffn_overrides(value, LLM_FFN_EXPS_REGEX, params.tensor_buft_overrides);
         }
     ).set_env("LLAMA_ARG_N_CPU_MOE"));
+    add_opt(common_arg(
+        {"--moe-cache-mib"}, "N",
+        "GPU cache size in MiB for the MoE experts kept in the CPU (default: 0, disabled)",
+        [](common_params & params, int value) {
+            if (value < 0) {
+                throw std::invalid_argument("invalid value");
+            }
+            params.moe_cache_size = (size_t) value*1024*1024;
+        }
+    ).set_env("LLAMA_ARG_MOE_CACHE_MIB"));
     add_opt(common_arg(
         {"-ncffn", "--n-cpu-ffn"}, "N",
         "keep the dense FFN weights of the first N layers in the CPU\n"
diff --git a/common/common.cpp b/common/common.cpp
index e36f8ab50..768b2e9a5 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1722,6 +1722,8 @@ struct llama_context_params common_context_params_to_llama(const common_params &
     cparams.type_k = params.cache_type_k;
     cparams.type_v = params.cache_type_v;

+    cparams.moe_cache_size = params.moe_cache_size;
+
     return cparams;
 }

diff --git a/common/common.h b/common/common.h
index 3e3eff379..0f912d901 100644
--- a/common/common.h
+++ b/common/common.h
@@ -593,6 +593,8 @@ struct common_params {
     ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K
     ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V

+    size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU
+
     common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;

     // multimodal models (see tools/mtmd)
diff --git a/common/speculative.cpp b/common/speculative.cpp
index a7f095a12..d9ddf44d2 100644
--- a/common/speculative.cpp
+++ b/common/speculative.cpp
@@ -2561,6 +2561,9 @@ common_params common_base_params_to_speculative(const common_params & params) {
     result.n_outputs_max = params.n_parallel;
     result.n_outputs_max_per_seq = 1;

+    // the MoE cache is only used by the target context
+    result.moe_cache_size = 0;
+
     // dflash/dspark decode the whole noise block in a single pass and sample every block position on the backend
     // TODO: refactor such properties to be announced by the speculative types
     //       something like `struct common_speculative_type_props common_speculative_type_get_props(...);`
diff --git a/include/llama.h b/include/llama.h
index 260247e82..77f527d5b 100644
--- a/include/llama.h
+++ b/include/llama.h
@@ -396,6 +396,8 @@ extern "C" {
         enum ggml_type type_k; // data type for K cache [EXPERIMENTAL]
         enum ggml_type type_v; // data type for V cache [EXPERIMENTAL]

+        size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, 0 = disabled [EXPERIMENTAL]
+
         // Abort callback
         // if it returns true, execution of llama_decode() will be aborted
         // currently works only with CPU execution
diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt
index afdaddc79..97250c2e8 100644
--- a/src/CMakeLists.txt
+++ b/src/CMakeLists.txt
@@ -28,6 +28,7 @@ set(LLAMA_CORE_SOURCES
     llama-kv-cache-msa.cpp
     llama-kv-cache-dsv4.cpp
     llama-memory.cpp
+    llama-moe-cache.cpp
     llama-memory-hybrid.cpp
     llama-memory-hybrid-iswa.cpp
     llama-memory-hybrid-idx.cpp
diff --git a/src/llama-context.cpp b/src/llama-context.cpp
index ff2ea461c..5f861f96c 100644
--- a/src/llama-context.cpp
+++ b/src/llama-context.cpp
@@ -9,6 +9,7 @@
 #include "llama-memory.h"
 #include "llama-mmap.h"
 #include "llama-model.h"
+#include "llama-moe-cache.h"
 #include "llama-ext.h"
 #include "llama-sampler.h"
 #include "llama.h"
@@ -272,8 +273,9 @@ llama_context::llama_context(
         }
     }

-    cparams.op_offload = params.op_offload;
-    cparams.kv_unified = params.kv_unified;
+    cparams.op_offload     = params.op_offload;
+    cparams.kv_unified     = params.kv_unified;
+    cparams.moe_cache_size = params.moe_cache_size;

     // initialized later
     cparams.pipeline_parallel = false;
@@ -462,6 +464,22 @@ llama_context::llama_context(
             LLAMA_LOG_INFO("%s: pipeline parallelism enabled\n", __func__);
         }

+        if (cparams.moe_cache_size > 0) {
+            if (cparams.pipeline_parallel || model.n_devices() > 1) {
+                throw std::runtime_error("MoE cache does not support multiple devices");
+            }
+            for (size_t i = 0; i < backend_ptrs.size(); ++i) {
+                const auto type = ggml_backend_dev_type(ggml_backend_get_device(backend_ptrs[i]));
+                if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
+                    moe_cache = std::make_unique<llama_moe_cache>(model, backend_ptrs[i], backend_buft[i], cparams.moe_cache_size);
+                    break;
+                }
+            }
+            if (!moe_cache) {
+                throw std::runtime_error("MoE cache requires a GPU backend");
+            }
+        }
+
         sched_reserve();

         if (!cparams.flash_attn) {
@@ -2605,6 +2623,7 @@ llm_graph_params llama_context::graph_params(
         /*.loras       =*/ loras.get(),
         /*.mctx        =*/ mctx,
         /*.cross       =*/ &cross,
+        /*.moe_cache   =*/ moe_cache.get(),
         /*.prec_policy =*/ &model.prec_policy,
         /*.samplers    =*/ sampling.samplers,
         /*.n_outputs   =*/ n_outputs,
@@ -2645,7 +2664,14 @@ ggml_status llama_context::graph_compute(
 }

 bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data) {
-    auto & st = static_cast<llama_context *>(user_data)->copy_experts;
+    auto * lctx = static_cast<llama_context *>(user_data);
+
+    // the slot maps of the MoE cache
+    if (lctx->moe_cache && lctx->moe_cache->copy(backend, src, dst, graph)) {
+        return true;
+    }
+
+    auto & st = lctx->copy_experts;

     // the ids must be computed before the split starts, so only the first node of the split is considered
     if (ggml_graph_n_nodes(graph) == 0) {
@@ -2692,10 +2718,28 @@ bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor
             last++;
         }

+        // the experts in the MoE cache are copied from device memory, the others are uploaded
+        int64_t next = first;
+        for (int64_t e = first; e <= last && lctx->moe_cache; ) {
+            const int64_t n = lctx->moe_cache->copy_experts(backend, src, dst, e, last);
+            if (n == 0) {
+                e++;
+                continue;
+            }
+            if (next < e) {
+                ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + next*expert_size, next*expert_size, (e - next)*expert_size);
+            }
+            e   += n;
+            next = e;
+        }
+
         // copy a bit extra to ensure there are no NaNs in the padding of the last expert, this is necessary for MMQ in the CUDA backend
-        const size_t offset  = first*expert_size;
+        const size_t offset  = next*expert_size;
         const size_t padding = last < n_expert - 1 ? std::min<size_t>(expert_size, 512) : 0;
-        ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, (last - first + 1)*expert_size + padding);
+        const size_t size    = (last + 1 - next)*expert_size + padding;
+        if (size > 0) {
+            ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, size);
+        }

         first = last + 1;
     }
@@ -3562,6 +3606,11 @@ llama_memory_breakdown llama_context::memory_breakdown() const {
             ret[buft].context += size;
         }
     }
+    if (moe_cache) {
+        for (const auto & [buft, size] : moe_cache->memory_breakdown()) {
+            ret[buft].context += size;
+        }
+    }
     if (model.hparams.no_alloc) {
         for (size_t i = 0; i < backends.size(); ++i) {
             ggml_backend_t             backend = backends[i].get();
@@ -3851,6 +3900,7 @@ llama_context_params llama_context_default_params() {
         /*.cb_eval_user_data           =*/ nullptr,
         /*.type_k                      =*/ GGML_TYPE_F16,
         /*.type_v                      =*/ GGML_TYPE_F16,
+        /*.moe_cache_size              =*/ 0,
         /*.abort_callback              =*/ nullptr,
         /*.abort_callback_data         =*/ nullptr,
         /*.embeddings                  =*/ false,
diff --git a/src/llama-context.h b/src/llama-context.h
index 28d386ad7..69bc01c19 100644
--- a/src/llama-context.h
+++ b/src/llama-context.h
@@ -7,6 +7,7 @@
 #include "llama-adapter.h"
 #include "llama-impl.h"
 #include "llama-memory.h"
+#include "llama-moe-cache.h"

 #include "ggml-cpp.h"
 #include "ggml-opt.h"
@@ -17,6 +18,7 @@

 struct llama_model;
 class llama_batch_allocr;
+class llama_moe_cache;

 class llama_io_read_i;
 class llama_io_write_i;
@@ -271,7 +273,7 @@ private:

     llm_graph_cb graph_get_cb() const;

-    // ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID
+    // ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID and updates the MoE cache
     static bool sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data);

     // disable auto fused ops (Flash Attention, Gated Delta Net) whose op lands on a device
@@ -299,6 +301,7 @@ private:
     llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably

     llama_memory_ptr memory;
+    llama_moe_cache_ptr moe_cache;

     // decode output (2-dimensional array: [n_outputs][n_vocab])
     buffer_view<float> logits = {nullptr, 0};
diff --git a/src/llama-cparams.h b/src/llama-cparams.h
index 004fec5a6..f4eb181a7 100644
--- a/src/llama-cparams.h
+++ b/src/llama-cparams.h
@@ -55,6 +55,8 @@ struct llama_cparams {
     bool pipeline_parallel;
     bool training;           // set by llama_opt_init()

+    size_t moe_cache_size;
+
     std::vector<bool> embeddings_layer_inp; // [n_layer()] extract input embeddings for layer

     enum llama_context_type ctx_type;
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index 1112ad885..e0b7a47a9 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -2,6 +2,7 @@

 #include "llama-impl.h"
 #include "llama-model.h"
+#include "llama-moe-cache.h"
 #include "llama-batch.h"
 #include "llama-cparams.h"
 #include "llama-sampler.h"
@@ -1523,6 +1524,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) :
     loras            (params.loras),
     mctx             (params.mctx),
     cross            (params.cross),
+    moe_cache        (params.moe_cache),
     prec_policy      (params.prec_policy),
     samplers         (params.samplers),
     cb_func          (params.cb),
@@ -1589,8 +1591,12 @@ ggml_tensor * llm_graph_context::build_lora_mm_id(
           ggml_tensor * w,   // ggml_tensor * as
           ggml_tensor * cur, // ggml_tensor * b
           ggml_tensor * ids,
-          ggml_tensor * w_s) const {
-    ggml_tensor * res = ggml_mul_mat_id(ctx0, w, cur, ids);
+          ggml_tensor * w_s,
+          ggml_tensor * slots) const {
+    // the experts in the MoE cache are selected by their slots
+    ggml_tensor * res = slots == nullptr ?
+        ggml_mul_mat_id(ctx0, w, cur, ids) :
+        ggml_mul_mat_id(ctx0, moe_cache->get_experts(w), cur, slots);

     if (prec_policy) {
         prec_policy->apply(res);
@@ -2205,6 +2211,9 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
     //call early so that topk-moe can be used
     ggml_build_forward_expand(gf, weights);

+    // the experts of host-resident layers may be read from the MoE cache
+    ggml_tensor * slots = build_moe_cache_slots(selected_experts, up_exps, gate_exps, down_exps, gate_up_exps, il);
+
     cur = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);

     if (weight_before_ffn) {
@@ -2219,7 +2228,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(

     if (gate_up_exps) {
         // merged gate_up path: one mul_mat_id, then split into gate and up views
-        ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s); // [n_ff*2, n_expert_used, n_tokens]
+        ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff*2, n_expert_used, n_tokens]
         cb(gate_up, "ffn_moe_gate_up", il);

         if (up_exps_s) {
@@ -2238,7 +2247,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
         cb(up, "ffn_moe_up", il);
     } else {
         // separate gate and up path
-        up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s); // [n_ff, n_expert_used, n_tokens]
+        up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff, n_expert_used, n_tokens]
         cb(up, "ffn_moe_up", il);

         if (up_exps_s) {
@@ -2251,7 +2260,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
         }

         if (gate_exps) {
-            cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s); // [n_ff, n_expert_used, n_tokens]
+            cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s, slots); // [n_ff, n_expert_used, n_tokens]
             cb(cur, "ffn_moe_gate", il);
         } else {
             cur = up;
@@ -2352,7 +2361,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
             GGML_ABORT("fatal error");
     }

-    experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s); // [n_embd, n_expert_used, n_tokens]
+    experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s, slots); // [n_embd, n_expert_used, n_tokens]
     if (arch == LLM_ARCH_MISTRAL4) {
         // src1 can exceed F16 range
         ggml_prec_set_src(experts, GGML_PREC_F32, 1);
@@ -2409,6 +2418,45 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
     return moe_out;
 }

+ggml_tensor * llm_graph_context::build_moe_cache_slots(
+         ggml_tensor * selected_experts,
+         ggml_tensor * up_exps,
+         ggml_tensor * gate_exps,
+         ggml_tensor * down_exps,
+         ggml_tensor * gate_up_exps,
+                 int   il) const {
+    if (moe_cache == nullptr) {
+        return nullptr;
+    }
+
+    ggml_tensor * slot_map = moe_cache->get_slot_map(il, selected_experts->ne[1], selected_experts->ne[0]);
+    if (slot_map == nullptr) {
+        return nullptr;
+    }
+    for (ggml_tensor * w : { up_exps, gate_exps, down_exps, gate_up_exps }) {
+        if (w != nullptr && moe_cache->get_experts(w) == nullptr) {
+            return nullptr;
+        }
+    }
+
+    ggml_tensor * ids = selected_experts;
+    if (!ggml_is_contiguous(ids)) {
+        ids = ggml_cont(ctx0, ids);
+    }
+    ids = ggml_reshape_1d(ctx0, ids, ggml_nelements(ids));
+
+    // the slot map is a host weight, so the scheduler starts a new split here and copies it with the copy callback
+    // the callback reads the selected experts, uploads the missing ones and updates the slot map
+    ggml_tensor * slots = ggml_get_rows(ctx0, slot_map, ids); // [1, n_expert_used*n_tokens]
+    if (!ggml_backend_supports_op(moe_cache->backend(), slots)) {
+        return nullptr;
+    }
+    ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend());
+    cb(slots, "ffn_moe_slots", il);
+
+    return ggml_reshape_2d(ctx0, slots, selected_experts->ne[0], selected_experts->ne[1]); // [n_expert_used, n_tokens]
+}
+
 // input embeddings with optional lora
 ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float tok_scale) const {
     const int64_t n_embd_inp = hparams.n_embd_inp();
diff --git a/src/llama-graph.h b/src/llama-graph.h
index 838544576..79e8409ac 100644
--- a/src/llama-graph.h
+++ b/src/llama-graph.h
@@ -21,6 +21,8 @@ struct llama_cparams;
 struct llama_layer;
 struct llama_prec_policy;

+class llama_moe_cache;
+
 struct llama_memory_context_i;

 class llama_kv_cache_context;
@@ -793,6 +795,7 @@ struct llm_graph_params {
     const llama_adapter_loras    * loras;
     const llama_memory_context_i * mctx;
     const llama_cross            * cross;
+    const llama_moe_cache        * moe_cache;

     const llama_prec_policy * prec_policy = nullptr;

@@ -1036,6 +1039,7 @@ struct llm_graph_context {
     const llama_adapter_loras    * loras;
     const llama_memory_context_i * mctx;
     const llama_cross            * cross;
+    const llama_moe_cache        * moe_cache;

     const llama_prec_policy * prec_policy;

@@ -1078,11 +1082,13 @@ struct llm_graph_context {
               ggml_tensor * w_s = nullptr) const;

     // do mat_mul_id, while optionally apply lora and per-expert scale
+    // if slots is set, the experts are read from the MoE cache at these slots (see build_moe_cache_slots)
     ggml_tensor * build_lora_mm_id(
               ggml_tensor * w,   // ggml_tensor * as
               ggml_tensor * cur, // ggml_tensor * b
               ggml_tensor * ids,
-              ggml_tensor * w_s = nullptr) const;
+              ggml_tensor * w_s   = nullptr,
+              ggml_tensor * slots = nullptr) const;

     ggml_tensor * build_norm(
              ggml_tensor * cur,
@@ -1179,6 +1185,15 @@ struct llm_graph_context {
              ggml_tensor * down_exps_s = nullptr,
              ggml_tensor * selected_experts_in = nullptr) const;

+    // the slots of the selected experts in the MoE cache, nullptr if the experts of the layer are not read from the cache
+    ggml_tensor * build_moe_cache_slots(
+             ggml_tensor * selected_experts,
+             ggml_tensor * up_exps,
+             ggml_tensor * gate_exps,
+             ggml_tensor * down_exps,
+             ggml_tensor * gate_up_exps,
+                     int   il) const;
+
     //
     // inputs
     //
diff --git a/src/llama-moe-cache.cpp b/src/llama-moe-cache.cpp
new file mode 100644
index 000000000..7292a1e57
--- /dev/null
+++ b/src/llama-moe-cache.cpp
@@ -0,0 +1,550 @@
+#include "llama-moe-cache.h"
+
+#include "llama-impl.h"
+#include "llama-model.h"
+
+#include "ggml-cpp.h"
+
+#include <algorithm>
+#include <stdexcept>
+#include <unordered_map>
+#include <vector>
+
+namespace {
+
+// LRU of the experts of a group of layers, the slot of each expert is kept in the slot map of its layer
+struct moe_cache_lru {
+    int32_t n_expert = 0;
+    int32_t n_slots  = 0;
+
+    std::vector<int32_t *> slot_map; // [n_layer] data of the slot maps, -1 if the expert is not cached
+    std::vector<int32_t>   key_of;   // [n_slots] il*n_expert + expert, -1 if empty
+
+    // doubly linked list of the slots, head is the least recently used
+    std::vector<int32_t> prev;
+    std::vector<int32_t> next;
+    int32_t head = -1;
+    int32_t tail = -1;
+
+    std::vector<uint32_t> seen; // [n_expert]
+    uint32_t seen_gen = 0;
+    std::vector<int32_t> uniq;
+
+    void init(int32_t n_layer, int32_t n_expert, int32_t n_slots) {
+        this->n_expert = n_expert;
+        this->n_slots  = n_slots;
+        slot_map.assign(n_layer, nullptr);
+        key_of.assign(n_slots, -1);
+        prev.resize(n_slots);
+        next.resize(n_slots);
+        for (int32_t s = 0; s < n_slots; ++s) {
+            prev[s] = s - 1;
+            next[s] = s + 1 < n_slots ? s + 1 : -1;
+        }
+        head = 0;
+        tail = n_slots - 1;
+        seen.assign(n_expert, 0);
+    }
+
+    // move slot s to the tail (most recently used)
+    void touch(int32_t s) {
+        if (s == tail) {
+            return;
+        }
+        if (prev[s] >= 0) {
+            next[prev[s]] = next[s];
+        } else {
+            head = next[s];
+        }
+        prev[next[s]] = prev[s];
+
+        prev[s] = tail;
+        next[s] = -1;
+        next[tail] = s;
+        tail = s;
+    }
+
+    struct fill {
+        int32_t expert;
+        int32_t slot;
+    };
+
+    // give a slot to each expert selected by ids in layer il, the misses evict the least recently used experts
+    // returns false if the ids select more distinct experts than there are slots
+    bool plan(int32_t il, const int32_t * ids, size_t n_ids, std::vector<fill> & fills, size_t & n_hit) {
+        fills.clear();
+        n_hit = 0;
+
+        if (++seen_gen == 0) {
+            std::fill(seen.begin(), seen.end(), 0);
+            seen_gen = 1;
+        }
+        uniq.clear();
+        for (size_t i = 0; i < n_ids; ++i) {
+            GGML_ASSERT(ids[i] >= 0 && ids[i] < n_expert);
+            if (seen[ids[i]] != seen_gen) {
+                seen[ids[i]] = seen_gen;
+                uniq.push_back(ids[i]);
+            }
+        }
+        if (uniq.size() > (size_t) n_slots) {
+            return false;
+        }
+
+        int32_t * slots = slot_map[il];
+
+        // hits go to the tail first, so the head can be evicted below
+        for (int32_t e : uniq) {
+            if (slots[e] >= 0) {
+                touch(slots[e]);
+                n_hit++;
+            }
+        }
+        // sorted misses usually get consecutive slots, so the uploads can be merged
+        std::sort(uniq.begin(), uniq.end());
+        for (int32_t e : uniq) {
+            if (slots[e] >= 0) {
+                continue;
+            }
+            const int32_t s = head;
+            if (key_of[s] >= 0) {
+                slot_map[key_of[s] / n_expert][key_of[s] % n_expert] = -1;
+            }
+            key_of[s] = il*n_expert + e;
+            slots[e] = s;
+            touch(s);
+            fills.push_back({ e, s });
+        }
+        return true;
+    }
+};
+
+// gate, up, down or gate_up, down
+static std::vector<ggml_tensor *> llama_moe_cache_layer_experts(const llama_layer & layer) {
+    std::vector<ggml_tensor *> res;
+    for (ggml_tensor * t : { layer.ffn_gate_up_exps, layer.ffn_gate_exps, layer.ffn_up_exps, layer.ffn_down_exps }) {
+        if (t != nullptr) {
+            res.push_back(t);
+        }
+    }
+    return res;
+}
+
+static bool llama_moe_cache_same_layout(const std::vector<ggml_tensor *> & a, const std::vector<ggml_tensor *> & b) {
+    if (a.size() != b.size()) {
+        return false;
+    }
+    for (size_t i = 0; i < a.size(); ++i) {
+        if (a[i]->type != b[i]->type || !ggml_are_same_shape(a[i], b[i]) || a[i]->nb[2] != b[i]->nb[2]) {
+            return false;
+        }
+    }
+    return true;
+}
+
+static bool llama_moe_cache_is_host_weight(const ggml_tensor * t) {
+    return t->buffer != nullptr &&
+        ggml_backend_buffer_get_usage(t->buffer) == GGML_BACKEND_BUFFER_USAGE_WEIGHTS &&
+        ggml_backend_buffer_is_host(t->buffer);
+}
+
+}
+
+struct llama_moe_cache::impl {
+    // layers with the same expert tensor layout share the banks and the LRU of a group
+    struct group {
+        std::vector<ggml_tensor *> ref;   // expert tensors of the first layer
+        std::vector<int32_t> layers;
+        std::vector<ggml_tensor *> banks; // device storage of all slots, one per expert tensor
+        size_t host_bytes = 0;
+        int32_t n_slots = 0;
+        moe_cache_lru lru;
+    };
+
+    struct layer {
+        int32_t ig = -1;                    // -1 if the layer is not cached
+        ggml_tensor * slot_map = nullptr;   // I32 [1, n_expert] in host memory
+        std::vector<ggml_tensor *> experts; // host expert tensors, in the order of the banks
+    };
+
+    struct binding {
+        int32_t il;
+        int32_t ip;           // index of the bank
+        ggml_tensor * cached; // view of the bank used in place of the host experts
+    };
+
+    struct stats {
+        size_t hits   = 0;
+        size_t misses = 0;
+        size_t bytes  = 0;
+    };
+
+    static constexpr int64_t max_batch = 32;
+
+    ggml_backend_t backend;
+    int32_t n_expert_used;
+
+    stats stats_small; // up to 8 tokens per ubatch
+    stats stats_large;
+    stats stats_copy;  // experts copied from the cache for large batches
+
+    std::vector<group> groups;
+    std::vector<layer> layers;
+    std::unordered_map<const ggml_tensor *, binding> bindings; // host experts -> cached experts
+    std::unordered_map<const ggml_tensor *, int32_t> layer_of; // slot map -> layer
+
+    std::vector<int32_t> ids;
+    std::vector<moe_cache_lru::fill> fills;
+
+    // banks and their views on the device
+    ggml_context_ptr ctx;
+    ggml_backend_buffer_ptr buf;
+    size_t buf_size = 0;
+
+    // slot maps in host memory
+    ggml_context_ptr ctx_host;
+    ggml_backend_buffer_ptr buf_host;
+    size_t buf_host_size = 0;
+
+    // views used by copy_experts
+    ggml_context_ptr ctx_views;
+
+    impl(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
+            backend(backend), n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) {
+        ggml_backend_dev_t dev = ggml_backend_get_device(backend);
+        const auto dev_type = ggml_backend_dev_type(dev);
+        if (dev_type != GGML_BACKEND_DEVICE_TYPE_GPU && dev_type != GGML_BACKEND_DEVICE_TYPE_IGPU) {
+            throw std::runtime_error("MoE cache requires a GPU backend");
+        }
+        if (model.split_mode() == LLAMA_SPLIT_MODE_TENSOR) {
+            throw std::runtime_error("MoE cache does not support tensor parallelism");
+        }
+        if (model.hparams.n_expert == 0 || n_expert_used == 0) {
+            throw std::runtime_error("MoE cache requires a MoE model");
+        }
+
+        // only cache layers that keep all of their experts in host memory
+        size_t host_bytes = 0;
+        for (size_t il = 0; il < model.layers.size(); ++il) {
+            auto experts = llama_moe_cache_layer_experts(model.layers[il]);
+            if (experts.empty() || model.dev_layer(il) != dev ||
+                !std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) {
+                continue;
+            }
+            auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return llama_moe_cache_same_layout(g.ref, experts); });
+            if (it == groups.end()) {
+                groups.emplace_back();
+                it = groups.end() - 1;
+                it->ref = experts;
+            }
+            it->layers.push_back(il);
+            for (const ggml_tensor * t : experts) {
+                it->host_bytes += ggml_nbytes(t);
+                host_bytes     += ggml_nbytes(t);
+            }
+        }
+        if (groups.empty()) {
+            LLAMA_LOG_WARN("%s: no layer has all of its experts in host memory, MoE cache is disabled\n", __func__);
+            return;
+        }
+
+        // one extra slot at the end, CUDA MMQ can read past the last expert
+        const size_t alignment = ggml_backend_buft_get_alignment(buft);
+        auto alloc_size = [&](const group & g, int32_t n_slots) {
+            size_t res = 0;
+            for (const ggml_tensor * t : g.ref) {
+                res += GGML_PAD(t->nb[2]*(n_slots + 1), alignment);
+            }
+            return res;
+        };
+
+        // split the budget by the size of the experts, so each group caches the same fraction of its experts
+        size_t n_tensors      = 0;
+        size_t n_tensors_host = 0;
+        for (group & g : groups) {
+            const int32_t n_expert  = g.ref[0]->ne[2];
+            const size_t  budget    = (size_t) ((double) size*g.host_bytes/host_bytes);
+            const int32_t max_slots = g.layers.size()*n_expert;
+            while (g.n_slots < max_slots && alloc_size(g, g.n_slots + 1) <= budget) {
+                g.n_slots++;
+            }
+            if (g.n_slots < n_expert_used) {
+                LLAMA_LOG_WARN("%s: MoE cache budget is too small for %zu layers, they are not cached\n", __func__, g.layers.size());
+                g.n_slots = 0;
+                continue;
+            }
+            g.lru.init(model.layers.size(), n_expert, g.n_slots);
+            n_tensors      += g.ref.size()*(1 + g.layers.size());
+            n_tensors_host += g.layers.size();
+        }
+        if (n_tensors == 0) {
+            throw std::runtime_error("MoE cache is too small to hold the experts of one token");
+        }
+
+        auto init_ctx = [](size_t n_tensors) {
+            ggml_init_params params = {
+                /*.mem_size   =*/ n_tensors*ggml_tensor_overhead(),
+                /*.mem_buffer =*/ nullptr,
+                /*.no_alloc   =*/ true,
+            };
+            ggml_context_ptr res(ggml_init(params));
+            if (!res) {
+                throw std::runtime_error("failed to create the MoE cache context");
+            }
+            return res;
+        };
+        ctx       = init_ctx(n_tensors);
+        ctx_host  = init_ctx(n_tensors_host);
+        ctx_views = init_ctx(2);
+
+        ggml_backend_buffer_type_t buft_host = ggml_backend_cpu_buffer_type();
+        const size_t alignment_host = ggml_backend_buft_get_alignment(buft_host);
+
+        for (size_t ig = 0; ig < groups.size(); ++ig) {
+            group & g = groups[ig];
+            if (g.n_slots == 0) {
+                continue;
+            }
+            for (const ggml_tensor * t : g.ref) {
+                ggml_tensor * bank = ggml_new_tensor_3d(ctx.get(), t->type, t->ne[0], t->ne[1], g.n_slots + 1);
+                GGML_ASSERT(bank->nb[2] == t->nb[2]);
+                ggml_format_name(bank, "moe_cache.%zu.%s", ig, t->name);
+                g.banks.push_back(bank);
+            }
+            for (int32_t il : g.layers) {
+                layer & l = layers[il];
+                l.ig      = (int32_t) ig;
+                l.experts = llama_moe_cache_layer_experts(model.layers[il]);
+                for (size_t ip = 0; ip < l.experts.size(); ++ip) {
+                    ggml_tensor * bank   = g.banks[ip];
+                    ggml_tensor * cached = ggml_view_3d(ctx.get(), bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0);
+                    ggml_format_name(cached, "moe_cache.%s", l.experts[ip]->name);
+                    bindings[l.experts[ip]] = { il, (int32_t) ip, cached };
+                }
+                l.slot_map = ggml_new_tensor_2d(ctx_host.get(), GGML_TYPE_I32, 1, g.ref[0]->ne[2]);
+                ggml_format_name(l.slot_map, "moe_cache.slot_map-%d", il);
+                layer_of[l.slot_map] = il;
+                buf_host_size += GGML_PAD(ggml_nbytes(l.slot_map), alignment_host);
+            }
+            buf_size += alloc_size(g, g.n_slots);
+        }
+
+        if (model.hparams.no_alloc) {
+            // only used to measure the memory use, see llama_context::memory_breakdown
+            buf.reset(ggml_backend_buft_alloc_buffer(buft, 0));
+            buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0));
+            for (ggml_tensor * t = ggml_get_first_tensor(ctx.get()); t != nullptr; t = ggml_get_next_tensor(ctx.get(), t)) {
+                t->buffer = buf.get();
+            }
+            for (ggml_tensor * t = ggml_get_first_tensor(ctx_host.get()); t != nullptr; t = ggml_get_next_tensor(ctx_host.get(), t)) {
+                t->buffer = buf_host.get();
+            }
+        } else {
+            buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx.get(), buft));
+            buf_host.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx_host.get(), buft_host));
+            if (!buf || !buf_host) {
+                throw std::runtime_error("failed to allocate the MoE cache buffers");
+            }
+            ggml_backend_buffer_clear(buf.get(), 0);
+            ggml_backend_buffer_clear(buf_host.get(), 0xff); // all slots are -1
+            buf_size      = ggml_backend_buffer_get_size(buf.get());
+            buf_host_size = ggml_backend_buffer_get_size(buf_host.get());
+
+            for (group & g : groups) {
+                for (int32_t il : g.layers) {
+                    if (layers[il].slot_map != nullptr) {
+                        g.lru.slot_map[il] = (int32_t *) layers[il].slot_map->data;
+                    }
+                }
+            }
+        }
+
+        // as weights, the ops that read the banks run on the device and the slot maps are copied with the copy callback
+        ggml_backend_buffer_set_usage(buf.get(),      GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
+        ggml_backend_buffer_set_usage(buf_host.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
+
+        LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__,
+            ggml_backend_buft_name(buft), buf_size/1024.0/1024.0, host_bytes/1024.0/1024.0);
+        for (const group & g : groups) {
+            LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__,
+                g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2]));
+        }
+    }
+
+    ~impl() {
+        log_stats();
+    }
+
+    ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
+        if (il < 0 || il >= (int32_t) layers.size() || layers[il].ig < 0) {
+            return nullptr;
+        }
+        const layer & l = layers[il];
+
+        // large batches use most experts of a layer, so they gain little from the cache and would evict the experts used in generation
+        if (n_tokens == 0 || n_tokens > max_batch || std::min(n_tokens*n_expert_used, l.slot_map->ne[1]) > groups[l.ig].n_slots) {
+            return nullptr;
+        }
+        return l.slot_map;
+    }
+
+    ggml_tensor * get_experts(const ggml_tensor * w) const {
+        const auto it = bindings.find(w);
+        return it != bindings.end() ? it->second.cached : nullptr;
+    }
+
+    int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
+        const auto it = bindings.find(w);
+        if (it == bindings.end() || backend != this->backend) {
+            return 0;
+        }
+        const binding & b = it->second;
+        const group & g = groups[layers[b.il].ig];
+
+        // large batches only read the cache, so the experts used in generation stay in it
+        const int32_t * slots = g.lru.slot_map[b.il];
+        if (slots == nullptr || slots[e] < 0) {
+            return 0;
+        }
+        int64_t n = 1;
+        while (e + n <= last && slots[e + n] == slots[e] + n) {
+            n++;
+        }
+
+        ggml_tensor * bank = g.banks[b.ip];
+        ggml_reset(ctx_views.get());
+        ggml_tensor * src_view = ggml_view_3d(ctx_views.get(), bank, bank->ne[0], bank->ne[1], n, bank->nb[1], bank->nb[2], slots[e]*bank->nb[2]);
+        ggml_tensor * dst_view = ggml_view_3d(ctx_views.get(), dst,  dst->ne[0],  dst->ne[1],  n, dst->nb[1],  dst->nb[2],  e*dst->nb[2]);
+        ggml_backend_view_init(src_view);
+        ggml_backend_view_init(dst_view);
+        ggml_backend_tensor_copy_async(backend, backend, src_view, dst_view);
+
+        stats_copy.hits  += n;
+        stats_copy.bytes += ggml_nbytes(src_view);
+
+        return n;
+    }
+
+    bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) {
+        const auto it = layer_of.find(src);
+        if (it == layer_of.end()) {
+            return false;
+        }
+        const int32_t il = it->second;
+        const layer & l = layers[il];
+        group & g = groups[l.ig];
+
+        GGML_ASSERT(backend == this->backend);
+
+        // the get_rows that looks up the slots of the selected experts
+        const int n_nodes = ggml_graph_n_nodes(graph);
+        const ggml_tensor * lookup = nullptr;
+        for (int i = 0; i < n_nodes && lookup == nullptr; ++i) {
+            const ggml_tensor * node = ggml_graph_node(graph, i);
+            if (node->op == GGML_OP_GET_ROWS && node->src[0] == dst) {
+                lookup = node;
+            }
+        }
+        GGML_ASSERT(lookup != nullptr);
+
+        // the selected experts must be computed in an earlier split
+        // the scheduler starts a new split at the lookup because it reads a host weight, but only if the split already has inputs
+        const ggml_tensor * sel = lookup->src[1];
+        for (int i = 0; i < n_nodes; ++i) {
+            const ggml_tensor * node = ggml_graph_node(graph, i);
+            if (node == sel || node == sel->view_src) {
+                GGML_ABORT("the experts of layer %d are selected in the same split as their MoE cache lookup", il);
+            }
+        }
+        GGML_ASSERT(ggml_is_contiguous(sel));
+
+        ids.resize(ggml_nelements(sel));
+        ggml_backend_tensor_get_async(backend, sel, ids.data(), 0, ggml_nbytes(sel));
+        ggml_backend_synchronize(backend);
+
+        size_t n_hit = 0;
+        if (!g.lru.plan(il, ids.data(), ids.size(), fills, n_hit)) {
+            GGML_ABORT("the MoE cache is too small for the experts selected in layer %d", il);
+        }
+
+        // upload the missing experts, consecutive experts going to consecutive slots are uploaded together
+        size_t bytes = 0;
+        for (size_t ip = 0; ip < l.experts.size(); ++ip) {
+            const ggml_tensor * w    = l.experts[ip];
+            ggml_tensor       * bank = g.banks[ip];
+            const size_t expert_size = w->nb[2];
+            for (size_t i = 0; i < fills.size();) {
+                size_t n = 1;
+                while (i + n < fills.size() && fills[i + n].expert == fills[i].expert + (int32_t) n && fills[i + n].slot == fills[i].slot + (int32_t) n) {
+                    n++;
+                }
+                ggml_backend_tensor_set_async(backend, bank, (const uint8_t *) w->data + fills[i].expert*expert_size, fills[i].slot*expert_size, n*expert_size);
+                bytes += n*expert_size;
+                i += n;
+            }
+        }
+
+        stats & st = ids.size() <= (size_t) 8*n_expert_used ? stats_small : stats_large;
+        st.hits   += n_hit;
+        st.misses += fills.size();
+        st.bytes  += bytes;
+
+        // the next copy synchronizes the backend before it changes the slot map again
+        ggml_backend_tensor_set_async(backend, dst, src->data, 0, ggml_nbytes(src));
+
+        return true;
+    }
+
+    void log_stats() const {
+        auto log = [](const char * name, const stats & st) {
+            const size_t n = st.hits + st.misses;
+            if (n == 0) {
+                return;
+            }
+            LLAMA_LOG_INFO("llama_moe_cache: %s: hits = %zu, misses = %zu, hit rate = %.2f%%, uploaded = %.2f MiB\n",
+                name, st.hits, st.misses, 100.0*st.hits/n, st.bytes/1024.0/1024.0);
+        };
+        log("ubatch <= 8", stats_small);
+        log("ubatch  > 8", stats_large);
+        if (stats_copy.hits > 0) {
+            LLAMA_LOG_INFO("llama_moe_cache: large batches: %zu experts copied from the cache, %.2f MiB\n", stats_copy.hits, stats_copy.bytes/1024.0/1024.0);
+        }
+    }
+};
+
+llama_moe_cache::llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
+    pimpl(new impl(model, backend, buft, size)) {
+}
+
+llama_moe_cache::~llama_moe_cache() = default;
+
+ggml_backend_t llama_moe_cache::backend() const {
+    return pimpl->backend;
+}
+
+ggml_tensor * llama_moe_cache::get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
+    return pimpl->get_slot_map(il, n_tokens, n_expert_used);
+}
+
+ggml_tensor * llama_moe_cache::get_experts(const ggml_tensor * w) const {
+    return pimpl->get_experts(w);
+}
+
+bool llama_moe_cache::copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) {
+    return pimpl->copy(backend, src, dst, graph);
+}
+
+int64_t llama_moe_cache::copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
+    return pimpl->copy_experts(backend, w, dst, e, last);
+}
+
+std::map<ggml_backend_buffer_type_t, size_t> llama_moe_cache::memory_breakdown() const {
+    std::map<ggml_backend_buffer_type_t, size_t> res;
+    if (pimpl->buf) {
+        res[ggml_backend_buffer_get_type(pimpl->buf.get())] += pimpl->buf_size;
+    }
+    if (pimpl->buf_host) {
+        res[ggml_backend_buffer_get_type(pimpl->buf_host.get())] += pimpl->buf_host_size;
+    }
+    return res;
+}
diff --git a/src/llama-moe-cache.h b/src/llama-moe-cache.h
new file mode 100644
index 000000000..4872370d4
--- /dev/null
+++ b/src/llama-moe-cache.h
@@ -0,0 +1,39 @@
+#pragma once
+
+#include "ggml-backend.h"
+
+#include <map>
+#include <memory>
+
+struct llama_model;
+
+// keeps the most recently used experts of host-resident MoE layers in a device buffer
+// each layer has a slot map in host memory: when the scheduler copies it to the device, the copy callback uploads the missing experts
+class llama_moe_cache {
+public:
+    llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size);
+    ~llama_moe_cache();
+
+    ggml_backend_t backend() const;
+
+    // the slot map of layer il, if its experts can be read from the cache for n_tokens tokens, nullptr otherwise
+    ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const;
+
+    // the experts of w in the cache, nullptr if w is not cached
+    ggml_tensor * get_experts(const ggml_tensor * w) const;
+
+    // ggml_backend_sched copy callback, returns false if src is not a slot map
+    bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph);
+
+    // for large batches: copy the experts of w that are in the cache, starting at expert e and up to expert last, to the copy dst of w
+    // returns the number of experts copied, 0 if expert e is not in the cache
+    int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last);
+
+    std::map<ggml_backend_buffer_type_t, size_t> memory_breakdown() const;
+
+private:
+    struct impl;
+    std::unique_ptr<impl> pimpl;
+};
+
+using llama_moe_cache_ptr = std::unique_ptr<llama_moe_cache>;
diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp
index 54122629a..b54b3cc18 100644
--- a/tests/test-llama-archs.cpp
+++ b/tests/test-llama-archs.cpp
@@ -495,7 +495,7 @@ static std::pair<llama_model_ptr, llama_context_ptr> get_model_and_ctx(
         struct gguf_context * gguf_ctx, FILE * file, const size_t seed, const float stdev,
         const std::vector<ggml_backend_dev_t> & devs,
         const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false,
-        const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr) {
+        const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr, const size_t moe_cache_size = 0) {
     GGML_ASSERT((gguf_ctx == nullptr) != (file == nullptr));
     llama_model_params model_params = llama_model_default_params();
     model_params.progress_callback = silent_model_load_progress;
@@ -512,6 +512,11 @@ static std::pair<llama_model_ptr, llama_context_ptr> get_model_and_ctx(
     if (!encode) {
         ctx_params.n_ubatch = 64;
     }
+    if (moe_cache_size > 0) {
+        // the MoE cache is only used for small ubatches
+        ctx_params.moe_cache_size = moe_cache_size;
+        ctx_params.n_ubatch = 2;
+    }

     tensor_data_params tensor_params = { seed, stdev };
     llama_model_ptr model(gguf_ctx != nullptr ?
@@ -865,9 +870,10 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
         std::string                     label;
         llama_split_mode                split_mode;
         bool                            host_experts; // keep the experts in host memory, see host_experts_test
+        size_t                          moe_cache_size;

-        device_config(std::vector<ggml_backend_dev_t> devs, std::string name, llama_split_mode split_mode, bool host_experts = false)
-            : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts) {}
+        device_config(std::vector<ggml_backend_dev_t> devs, std::string name, llama_split_mode split_mode, bool host_experts = false, size_t moe_cache_size = 0)
+            : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts), moe_cache_size(moe_cache_size) {}
     };

     const llama_model_tensor_buft_override host_experts_overrides[] = {
@@ -905,6 +911,16 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
             dev_configs.emplace_back(devices_meta, "Host experts", LLAMA_SPLIT_MODE_LAYER, true);
             max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
         }
+
+        // the ops that use the host experts run on a GPU and read the experts from a cache
+        // the cache has only a few slots (4 for 288 KiB experts), so the experts are evicted and uploaded again
+        if (!devices_meta.empty()) {
+            const enum ggml_backend_dev_type type = ggml_backend_dev_type(devices_meta[0]);
+            if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
+                dev_configs.emplace_back(std::vector<ggml_backend_dev_t>{devices_meta[0]}, "MoE cache", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024);
+                max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
+            }
+        }
     }

     size_t max_arch_name_length = 0;
@@ -987,7 +1003,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
                     }
                     if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) {
                         test_executed = true;
-                        model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides);
+                        model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size);
                         logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode);
                         const double nmse_val = nmse(logits_cpu, logits_dev);
                         snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val);
@@ -1053,7 +1069,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
                         ms.save(file);
                         rewind(file);

-                        auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides);
+                        auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size);
                         const std::vector<float> logits_roundtrip = get_logits(
                             model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode);
                         status_roundtrip = "\033[1;32mOK\033[0m";