Commit c811cb8f0 for llama.cpp
commit c811cb8f0ac91b8ac72a32f970bdd45037f20da7
Author: Aman Gupta <amangupta052@gmail.com>
Date: Thu Oct 8 15:50:49 2026 +0530
llama: support MoE cache over multiple GPUs (#30112)
diff --git a/common/arg.cpp b/common/arg.cpp
index 182f76f45..9efcf79ac 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -2778,7 +2778,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
).set_env("LLAMA_ARG_N_CPU_MOE"));
add_opt(common_arg(
{"--moe-cache-mib"}, "N",
- "GPU cache size in MiB for the MoE experts kept in the CPU (default: 0, disabled)",
+ "GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)",
[](common_params & params, int value) {
if (value < 0) {
throw std::invalid_argument("invalid value");
diff --git a/common/common.h b/common/common.h
index de88dfb9b..ef062bb84 100644
--- a/common/common.h
+++ b/common/common.h
@@ -593,7 +593,7 @@ struct common_params {
ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K
ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V
- size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU
+ size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU, split among the GPUs like the layers
common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;
diff --git a/include/llama.h b/include/llama.h
index 77f527d5b..60329024f 100644
--- a/include/llama.h
+++ b/include/llama.h
@@ -396,7 +396,7 @@ extern "C" {
enum ggml_type type_k; // data type for K cache [EXPERIMENTAL]
enum ggml_type type_v; // data type for V cache [EXPERIMENTAL]
- size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, 0 = disabled [EXPERIMENTAL]
+ size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, split among the devices like the layers, 0 = disabled [EXPERIMENTAL]
// Abort callback
// if it returns true, execution of llama_decode() will be aborted
diff --git a/src/llama-context.cpp b/src/llama-context.cpp
index 5f861f96c..9dcaf05a6 100644
--- a/src/llama-context.cpp
+++ b/src/llama-context.cpp
@@ -436,7 +436,8 @@ llama_context::llama_context(
model.n_gpu_layers() > model.hparams.n_layer_all &&
model.split_mode() == LLAMA_SPLIT_MODE_LAYER &&
cparams.offload_kqv &&
- !model.has_tensor_overrides();
+ !model.has_tensor_overrides() &&
+ cparams.moe_cache_size == 0; // not supported by the MoE cache
// pipeline parallelism requires support for async compute and events in all devices
if (pipeline_parallel) {
@@ -465,19 +466,7 @@ llama_context::llama_context(
}
if (cparams.moe_cache_size > 0) {
- if (cparams.pipeline_parallel || model.n_devices() > 1) {
- throw std::runtime_error("MoE cache does not support multiple devices");
- }
- for (size_t i = 0; i < backend_ptrs.size(); ++i) {
- const auto type = ggml_backend_dev_type(ggml_backend_get_device(backend_ptrs[i]));
- if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
- moe_cache = std::make_unique<llama_moe_cache>(model, backend_ptrs[i], backend_buft[i], cparams.moe_cache_size);
- break;
- }
- }
- if (!moe_cache) {
- throw std::runtime_error("MoE cache requires a GPU backend");
- }
+ moe_cache = std::make_unique<llama_moe_cache>(model, backend_ptrs, backend_buft, cparams.moe_cache_size);
}
sched_reserve();
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index f030eb9e6..23e4c9091 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -2448,10 +2448,10 @@ ggml_tensor * llm_graph_context::build_moe_cache_slots(
// the slot map is a host weight, so the scheduler starts a new split here and copies it with the copy callback
// the callback reads the selected experts, uploads the missing ones and updates the slot map
ggml_tensor * slots = ggml_get_rows(ctx0, slot_map, ids); // [1, n_expert_used*n_tokens]
- if (!ggml_backend_supports_op(moe_cache->backend(), slots)) {
+ if (!ggml_backend_supports_op(moe_cache->backend(il), slots)) {
return nullptr;
}
- ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend());
+ ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend(il));
cb(slots, "ffn_moe_slots", il);
return ggml_reshape_2d(ctx0, slots, selected_experts->ne[0], selected_experts->ne[1]); // [n_expert_used, n_tokens]
diff --git a/src/llama-moe-cache.cpp b/src/llama-moe-cache.cpp
index 7292a1e57..f22d986dc 100644
--- a/src/llama-moe-cache.cpp
+++ b/src/llama-moe-cache.cpp
@@ -151,8 +151,22 @@ static bool llama_moe_cache_is_host_weight(const ggml_tensor * t) {
}
struct llama_moe_cache::impl {
- // layers with the same expert tensor layout share the banks and the LRU of a group
+ // a GPU with its own budget and banks, it caches the layers assigned to it
+ struct device {
+ ggml_backend_t backend;
+ ggml_backend_buffer_type_t buft;
+ size_t host_bytes = 0; // host experts of the layers it caches
+ double split = 0.0; // share of the budget
+
+ // banks and their views
+ ggml_context_ptr ctx;
+ ggml_backend_buffer_ptr buf;
+ size_t buf_size = 0;
+ };
+
+ // layers of the same device with the same expert tensor layout share the banks and the LRU of a group
struct group {
+ int32_t id; // device
std::vector<ggml_tensor *> ref; // expert tensors of the first layer
std::vector<int32_t> layers;
std::vector<ggml_tensor *> banks; // device storage of all slots, one per expert tensor
@@ -181,13 +195,13 @@ struct llama_moe_cache::impl {
static constexpr int64_t max_batch = 32;
- ggml_backend_t backend;
int32_t n_expert_used;
stats stats_small; // up to 8 tokens per ubatch
stats stats_large;
stats stats_copy; // experts copied from the cache for large batches
+ std::vector<device> devices;
std::vector<group> groups;
std::vector<layer> layers;
std::unordered_map<const ggml_tensor *, binding> bindings; // host experts -> cached experts
@@ -196,11 +210,6 @@ struct llama_moe_cache::impl {
std::vector<int32_t> ids;
std::vector<moe_cache_lru::fill> fills;
- // banks and their views on the device
- ggml_context_ptr ctx;
- ggml_backend_buffer_ptr buf;
- size_t buf_size = 0;
-
// slot maps in host memory
ggml_context_ptr ctx_host;
ggml_backend_buffer_ptr buf_host;
@@ -209,11 +218,17 @@ struct llama_moe_cache::impl {
// views used by copy_experts
ggml_context_ptr ctx_views;
- impl(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
- backend(backend), n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) {
- ggml_backend_dev_t dev = ggml_backend_get_device(backend);
- const auto dev_type = ggml_backend_dev_type(dev);
- if (dev_type != GGML_BACKEND_DEVICE_TYPE_GPU && dev_type != GGML_BACKEND_DEVICE_TYPE_IGPU) {
+ impl(const llama_model & model, const std::vector<ggml_backend_t> & backends, const std::vector<ggml_backend_buffer_type_t> & bufts, size_t size) :
+ n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) {
+ for (size_t i = 0; i < backends.size(); ++i) {
+ const auto dev_type = ggml_backend_dev_type(ggml_backend_get_device(backends[i]));
+ if (dev_type == GGML_BACKEND_DEVICE_TYPE_GPU || dev_type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
+ auto & d = devices.emplace_back();
+ d.backend = backends[i];
+ d.buft = bufts[i];
+ }
+ }
+ if (devices.empty()) {
throw std::runtime_error("MoE cache requires a GPU backend");
}
if (model.split_mode() == LLAMA_SPLIT_MODE_TENSOR) {
@@ -223,24 +238,28 @@ struct llama_moe_cache::impl {
throw std::runtime_error("MoE cache requires a MoE model");
}
- // only cache layers that keep all of their experts in host memory
- size_t host_bytes = 0;
+ // only cache layers that keep all of their experts in host memory, on the device the layer is assigned to
for (size_t il = 0; il < model.layers.size(); ++il) {
auto experts = llama_moe_cache_layer_experts(model.layers[il]);
- if (experts.empty() || model.dev_layer(il) != dev ||
- !std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) {
+ if (experts.empty() || !std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) {
+ continue;
+ }
+ const auto it_dev = std::find_if(devices.begin(), devices.end(), [&](const device & d) { return ggml_backend_get_device(d.backend) == model.dev_layer(il); });
+ if (it_dev == devices.end()) {
continue;
}
- auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return llama_moe_cache_same_layout(g.ref, experts); });
+ const int32_t id = (int32_t) (it_dev - devices.begin());
+ auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return g.id == id && llama_moe_cache_same_layout(g.ref, experts); });
if (it == groups.end()) {
groups.emplace_back();
it = groups.end() - 1;
+ it->id = id;
it->ref = experts;
}
it->layers.push_back(il);
for (const ggml_tensor * t : experts) {
- it->host_bytes += ggml_nbytes(t);
- host_bytes += ggml_nbytes(t);
+ it->host_bytes += ggml_nbytes(t);
+ devices[id].host_bytes += ggml_nbytes(t);
}
}
if (groups.empty()) {
@@ -249,8 +268,8 @@ struct llama_moe_cache::impl {
}
// one extra slot at the end, CUDA MMQ can read past the last expert
- const size_t alignment = ggml_backend_buft_get_alignment(buft);
auto alloc_size = [&](const group & g, int32_t n_slots) {
+ const size_t alignment = ggml_backend_buft_get_alignment(devices[g.id].buft);
size_t res = 0;
for (const ggml_tensor * t : g.ref) {
res += GGML_PAD(t->nb[2]*(n_slots + 1), alignment);
@@ -258,12 +277,43 @@ struct llama_moe_cache::impl {
return res;
};
- // split the budget by the size of the experts, so each group caches the same fraction of its experts
- size_t n_tensors = 0;
+ // the budget is split among the devices with host experts like the layers, by the tensor split or by default by free memory
+ const float * tensor_split = model.tensor_split();
+ const bool split_by_free = tensor_split == nullptr ||
+ std::all_of(tensor_split, tensor_split + model.n_devices(), [](float x) { return x == 0.0f; });
+ double split_sum = 0.0;
+ for (device & d : devices) {
+ if (d.host_bytes == 0) {
+ continue;
+ }
+ ggml_backend_dev_t dev = ggml_backend_get_device(d.backend);
+ if (split_by_free) {
+ size_t free;
+ size_t total;
+ ggml_backend_dev_memory(dev, &free, &total);
+ d.split = (double) free;
+ } else {
+ const auto it = std::find_if(model.devices.begin(), model.devices.end(), [&](const llama_device & ld) { return ld.dev == dev; });
+ GGML_ASSERT(it != model.devices.end());
+ d.split = (double) tensor_split[it - model.devices.begin()];
+ }
+ split_sum += d.split;
+ }
+ if (split_sum == 0.0) {
+ // the devices do not report their free memory
+ for (device & d : devices) {
+ d.split = d.host_bytes > 0 ? 1.0 : 0.0;
+ split_sum += d.split;
+ }
+ }
+
+ // within a device the budget is split by the size of the experts, so each group caches the same fraction of its experts
+ std::vector<size_t> n_tensors(devices.size(), 0);
size_t n_tensors_host = 0;
for (group & g : groups) {
+ const device & d = devices[g.id];
const int32_t n_expert = g.ref[0]->ne[2];
- const size_t budget = (size_t) ((double) size*g.host_bytes/host_bytes);
+ const size_t budget = (size_t) ((double) size*d.split/split_sum*g.host_bytes/d.host_bytes);
const int32_t max_slots = g.layers.size()*n_expert;
while (g.n_slots < max_slots && alloc_size(g, g.n_slots + 1) <= budget) {
g.n_slots++;
@@ -274,10 +324,10 @@ struct llama_moe_cache::impl {
continue;
}
g.lru.init(model.layers.size(), n_expert, g.n_slots);
- n_tensors += g.ref.size()*(1 + g.layers.size());
- n_tensors_host += g.layers.size();
+ n_tensors[g.id] += g.ref.size()*(1 + g.layers.size());
+ n_tensors_host += g.layers.size();
}
- if (n_tensors == 0) {
+ if (n_tensors_host == 0) {
throw std::runtime_error("MoE cache is too small to hold the experts of one token");
}
@@ -293,7 +343,11 @@ struct llama_moe_cache::impl {
}
return res;
};
- ctx = init_ctx(n_tensors);
+ for (size_t id = 0; id < devices.size(); ++id) {
+ if (n_tensors[id] > 0) {
+ devices[id].ctx = init_ctx(n_tensors[id]);
+ }
+ }
ctx_host = init_ctx(n_tensors_host);
ctx_views = init_ctx(2);
@@ -305,8 +359,9 @@ struct llama_moe_cache::impl {
if (g.n_slots == 0) {
continue;
}
+ ggml_context * ctx = devices[g.id].ctx.get();
for (const ggml_tensor * t : g.ref) {
- ggml_tensor * bank = ggml_new_tensor_3d(ctx.get(), t->type, t->ne[0], t->ne[1], g.n_slots + 1);
+ ggml_tensor * bank = ggml_new_tensor_3d(ctx, t->type, t->ne[0], t->ne[1], g.n_slots + 1);
GGML_ASSERT(bank->nb[2] == t->nb[2]);
ggml_format_name(bank, "moe_cache.%zu.%s", ig, t->name);
g.banks.push_back(bank);
@@ -317,7 +372,7 @@ struct llama_moe_cache::impl {
l.experts = llama_moe_cache_layer_experts(model.layers[il]);
for (size_t ip = 0; ip < l.experts.size(); ++ip) {
ggml_tensor * bank = g.banks[ip];
- ggml_tensor * cached = ggml_view_3d(ctx.get(), bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0);
+ ggml_tensor * cached = ggml_view_3d(ctx, bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0);
ggml_format_name(cached, "moe_cache.%s", l.experts[ip]->name);
bindings[l.experts[ip]] = { il, (int32_t) ip, cached };
}
@@ -326,28 +381,41 @@ struct llama_moe_cache::impl {
layer_of[l.slot_map] = il;
buf_host_size += GGML_PAD(ggml_nbytes(l.slot_map), alignment_host);
}
- buf_size += alloc_size(g, g.n_slots);
+ devices[g.id].buf_size += alloc_size(g, g.n_slots);
}
if (model.hparams.no_alloc) {
// only used to measure the memory use, see llama_context::memory_breakdown
- buf.reset(ggml_backend_buft_alloc_buffer(buft, 0));
- buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0));
- for (ggml_tensor * t = ggml_get_first_tensor(ctx.get()); t != nullptr; t = ggml_get_next_tensor(ctx.get(), t)) {
- t->buffer = buf.get();
+ for (device & d : devices) {
+ if (!d.ctx) {
+ continue;
+ }
+ d.buf.reset(ggml_backend_buft_alloc_buffer(d.buft, 0));
+ for (ggml_tensor * t = ggml_get_first_tensor(d.ctx.get()); t != nullptr; t = ggml_get_next_tensor(d.ctx.get(), t)) {
+ t->buffer = d.buf.get();
+ }
}
+ buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0));
for (ggml_tensor * t = ggml_get_first_tensor(ctx_host.get()); t != nullptr; t = ggml_get_next_tensor(ctx_host.get(), t)) {
t->buffer = buf_host.get();
}
} else {
- buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx.get(), buft));
+ for (device & d : devices) {
+ if (!d.ctx) {
+ continue;
+ }
+ d.buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(d.ctx.get(), d.buft));
+ if (!d.buf) {
+ throw std::runtime_error("failed to allocate the MoE cache buffers");
+ }
+ ggml_backend_buffer_clear(d.buf.get(), 0);
+ d.buf_size = ggml_backend_buffer_get_size(d.buf.get());
+ }
buf_host.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx_host.get(), buft_host));
- if (!buf || !buf_host) {
+ if (!buf_host) {
throw std::runtime_error("failed to allocate the MoE cache buffers");
}
- ggml_backend_buffer_clear(buf.get(), 0);
ggml_backend_buffer_clear(buf_host.get(), 0xff); // all slots are -1
- buf_size = ggml_backend_buffer_get_size(buf.get());
buf_host_size = ggml_backend_buffer_get_size(buf_host.get());
for (group & g : groups) {
@@ -360,17 +428,33 @@ struct llama_moe_cache::impl {
}
// as weights, the ops that read the banks run on the device and the slot maps are copied with the copy callback
- ggml_backend_buffer_set_usage(buf.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
+ for (device & d : devices) {
+ if (d.buf) {
+ ggml_backend_buffer_set_usage(d.buf.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
+ }
+ }
ggml_backend_buffer_set_usage(buf_host.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
- LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__,
- ggml_backend_buft_name(buft), buf_size/1024.0/1024.0, host_bytes/1024.0/1024.0);
- for (const group & g : groups) {
- LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__,
- g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2]));
+ for (size_t id = 0; id < devices.size(); ++id) {
+ const device & d = devices[id];
+ if (d.host_bytes == 0) {
+ continue;
+ }
+ LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__,
+ ggml_backend_buft_name(d.buft), d.buf_size/1024.0/1024.0, d.host_bytes/1024.0/1024.0);
+ for (const group & g : groups) {
+ if (g.id == (int32_t) id) {
+ LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__,
+ g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2]));
+ }
+ }
}
}
+ ggml_backend_t backend(int32_t il) const {
+ return devices[groups[layers[il].ig].id].backend;
+ }
+
~impl() {
log_stats();
}
@@ -395,11 +479,14 @@ struct llama_moe_cache::impl {
int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
const auto it = bindings.find(w);
- if (it == bindings.end() || backend != this->backend) {
+ if (it == bindings.end()) {
return 0;
}
const binding & b = it->second;
const group & g = groups[layers[b.il].ig];
+ if (backend != devices[g.id].backend) {
+ return 0;
+ }
// large batches only read the cache, so the experts used in generation stay in it
const int32_t * slots = g.lru.slot_map[b.il];
@@ -434,7 +521,7 @@ struct llama_moe_cache::impl {
const layer & l = layers[il];
group & g = groups[l.ig];
- GGML_ASSERT(backend == this->backend);
+ GGML_ASSERT(backend == devices[g.id].backend);
// the get_rows that looks up the slots of the selected experts
const int n_nodes = ggml_graph_n_nodes(graph);
@@ -512,14 +599,14 @@ struct llama_moe_cache::impl {
}
};
-llama_moe_cache::llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
- pimpl(new impl(model, backend, buft, size)) {
+llama_moe_cache::llama_moe_cache(const llama_model & model, const std::vector<ggml_backend_t> & backends, const std::vector<ggml_backend_buffer_type_t> & bufts, size_t size) :
+ pimpl(new impl(model, backends, bufts, size)) {
}
llama_moe_cache::~llama_moe_cache() = default;
-ggml_backend_t llama_moe_cache::backend() const {
- return pimpl->backend;
+ggml_backend_t llama_moe_cache::backend(int32_t il) const {
+ return pimpl->backend(il);
}
ggml_tensor * llama_moe_cache::get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
@@ -540,8 +627,10 @@ int64_t llama_moe_cache::copy_experts(ggml_backend_t backend, const ggml_tensor
std::map<ggml_backend_buffer_type_t, size_t> llama_moe_cache::memory_breakdown() const {
std::map<ggml_backend_buffer_type_t, size_t> res;
- if (pimpl->buf) {
- res[ggml_backend_buffer_get_type(pimpl->buf.get())] += pimpl->buf_size;
+ for (const auto & d : pimpl->devices) {
+ if (d.buf) {
+ res[ggml_backend_buffer_get_type(d.buf.get())] += d.buf_size;
+ }
}
if (pimpl->buf_host) {
res[ggml_backend_buffer_get_type(pimpl->buf_host.get())] += pimpl->buf_host_size;
diff --git a/src/llama-moe-cache.h b/src/llama-moe-cache.h
index 4872370d4..3b6726b86 100644
--- a/src/llama-moe-cache.h
+++ b/src/llama-moe-cache.h
@@ -4,6 +4,7 @@
#include <map>
#include <memory>
+#include <vector>
struct llama_model;
@@ -11,10 +12,12 @@ struct llama_model;
// each layer has a slot map in host memory: when the scheduler copies it to the device, the copy callback uploads the missing experts
class llama_moe_cache {
public:
- llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size);
+ // backends are all the backends of the context, each GPU gets its own cache of the given size for the layers assigned to it
+ llama_moe_cache(const llama_model & model, const std::vector<ggml_backend_t> & backends, const std::vector<ggml_backend_buffer_type_t> & bufts, size_t size);
~llama_moe_cache();
- ggml_backend_t backend() const;
+ // the device that caches layer il
+ ggml_backend_t backend(int32_t il) const;
// the slot map of layer il, if its experts can be read from the cache for n_tokens tokens, nullptr otherwise
ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const;
diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp
index b54b3cc18..78128ebf0 100644
--- a/tests/test-llama-archs.cpp
+++ b/tests/test-llama-archs.cpp
@@ -919,6 +919,11 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
dev_configs.emplace_back(std::vector<ggml_backend_dev_t>{devices_meta[0]}, "MoE cache", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024);
max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
+ // each GPU caches the layers assigned to it
+ if (devices_meta.size() > 1) {
+ dev_configs.emplace_back(devices_meta, "MoE cache, layer split", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024);
+ max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
+ }
}
}
}
diff --git a/tools/cli/README.md b/tools/cli/README.md
index acecc74a6..20cf01c6b 100644
--- a/tools/cli/README.md
+++ b/tools/cli/README.md
@@ -63,6 +63,7 @@
| `-ot, --override-tensor <tensor name pattern>=<buffer type>,...` | override tensor buffer type<br/>(env: LLAMA_ARG_OVERRIDE_TENSOR) |
| `-cmoe, --cpu-moe` | keep all Mixture of Experts (MoE) weights in the CPU<br/>(env: LLAMA_ARG_CPU_MOE) |
| `-ncmoe, --n-cpu-moe N` | keep the Mixture of Experts (MoE) weights of the first N layers in the CPU<br/>(env: LLAMA_ARG_N_CPU_MOE) |
+| `--moe-cache-mib N` | GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)<br/>(env: LLAMA_ARG_MOE_CACHE_MIB) |
| `-ncffn, --n-cpu-ffn N` | keep the dense FFN weights of the first N layers in the CPU<br/>(dense models; for MoE expert weights use --n-cpu-moe)<br/>(env: LLAMA_ARG_N_CPU_FFN) |
| `-ngl, --gpu-layers, --n-gpu-layers N` | max. number of layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS) |
| `-sm, --split-mode {none,layer,row,tensor}` | how to split the model across multiple GPUs, one of:<br/>- none: use one GPU only<br/>- layer (default): split layers and KV across GPUs (pipelined)<br/>- row: split weight across GPUs by rows (parallelized)<br/>- tensor: split weights and KV across GPUs (parallelized, EXPERIMENTAL)<br/>(env: LLAMA_ARG_SPLIT_MODE) |
@@ -204,6 +205,7 @@
| `--spec-draft-p-split, --draft-p-split P` | speculative decoding split probability (default: 0.10)<br/>(env: LLAMA_ARG_SPEC_DRAFT_P_SPLIT) |
| `--spec-draft-p-min, --draft-p-min P` | minimum speculative decoding probability (greedy) (default: 0.00)<br/>(env: LLAMA_ARG_SPEC_DRAFT_P_MIN) |
| `--spec-draft-backend-sampling, --no-spec-draft-backend-sampling` | offload draft sampling to the backend (default: enabled)<br/>(env: LLAMA_ARG_SPEC_DRAFT_BACKEND_SAMPLING) |
+| `--spec-draft-sampling {greedy,probabilistic}` | how the draft is sampled: greedy takes its argmax, probabilistic samples it and has the target verify by rejection sampling (default: greedy)<br/>(env: LLAMA_ARG_SPEC_DRAFT_SAMPLING) |
| `--spec-draft-device, -devd, --device-draft <dev1,dev2,..>` | comma-separated list of devices to use for offloading the draft model (none = don't offload, default: follows --device)<br/>use --list-devices to see a list of available devices |
| `--spec-draft-ngl, -ngld, --gpu-layers-draft, --n-gpu-layers-draft N` | max. number of draft model layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS_DRAFT) |
| `--spec-draft-model, -md, --model-draft FNAME` | draft model for speculative decoding (default: unused)<br/>(env: LLAMA_ARG_SPEC_DRAFT_MODEL) |
diff --git a/tools/completion/README.md b/tools/completion/README.md
index e2ac0668c..c396609ab 100644
--- a/tools/completion/README.md
+++ b/tools/completion/README.md
@@ -146,6 +146,7 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
| `-ot, --override-tensor <tensor name pattern>=<buffer type>,...` | override tensor buffer type<br/>(env: LLAMA_ARG_OVERRIDE_TENSOR) |
| `-cmoe, --cpu-moe` | keep all Mixture of Experts (MoE) weights in the CPU<br/>(env: LLAMA_ARG_CPU_MOE) |
| `-ncmoe, --n-cpu-moe N` | keep the Mixture of Experts (MoE) weights of the first N layers in the CPU<br/>(env: LLAMA_ARG_N_CPU_MOE) |
+| `--moe-cache-mib N` | GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)<br/>(env: LLAMA_ARG_MOE_CACHE_MIB) |
| `-ncffn, --n-cpu-ffn N` | keep the dense FFN weights of the first N layers in the CPU<br/>(dense models; for MoE expert weights use --n-cpu-moe)<br/>(env: LLAMA_ARG_N_CPU_FFN) |
| `-ngl, --gpu-layers, --n-gpu-layers N` | max. number of layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS) |
| `-sm, --split-mode {none,layer,row,tensor}` | how to split the model across multiple GPUs, one of:<br/>- none: use one GPU only<br/>- layer (default): split layers and KV across GPUs (pipelined)<br/>- row: split weight across GPUs by rows (parallelized)<br/>- tensor: split weights and KV across GPUs (parallelized, EXPERIMENTAL)<br/>(env: LLAMA_ARG_SPLIT_MODE) |
diff --git a/tools/server/README.md b/tools/server/README.md
index 9231f06e8..08d6f17a4 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -80,6 +80,7 @@ For the full list of features, please refer to [server's changelog](https://gith
| `-ot, --override-tensor <tensor name pattern>=<buffer type>,...` | override tensor buffer type<br/>(env: LLAMA_ARG_OVERRIDE_TENSOR) |
| `-cmoe, --cpu-moe` | keep all Mixture of Experts (MoE) weights in the CPU<br/>(env: LLAMA_ARG_CPU_MOE) |
| `-ncmoe, --n-cpu-moe N` | keep the Mixture of Experts (MoE) weights of the first N layers in the CPU<br/>(env: LLAMA_ARG_N_CPU_MOE) |
+| `--moe-cache-mib N` | GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)<br/>(env: LLAMA_ARG_MOE_CACHE_MIB) |
| `-ncffn, --n-cpu-ffn N` | keep the dense FFN weights of the first N layers in the CPU<br/>(dense models; for MoE expert weights use --n-cpu-moe)<br/>(env: LLAMA_ARG_N_CPU_FFN) |
| `-ngl, --gpu-layers, --n-gpu-layers N` | max. number of layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS) |
| `-sm, --split-mode {none,layer,row,tensor}` | how to split the model across multiple GPUs, one of:<br/>- none: use one GPU only<br/>- layer (default): split layers and KV across GPUs (pipelined)<br/>- row: split weight across GPUs by rows (parallelized)<br/>- tensor: split weights and KV across GPUs (parallelized, EXPERIMENTAL)<br/>(env: LLAMA_ARG_SPLIT_MODE) |
@@ -265,6 +266,7 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--spec-draft-p-split, --draft-p-split P` | speculative decoding split probability (default: 0.10)<br/>(env: LLAMA_ARG_SPEC_DRAFT_P_SPLIT) |
| `--spec-draft-p-min, --draft-p-min P` | minimum speculative decoding probability (greedy) (default: 0.00)<br/>(env: LLAMA_ARG_SPEC_DRAFT_P_MIN) |
| `--spec-draft-backend-sampling, --no-spec-draft-backend-sampling` | offload draft sampling to the backend (default: enabled)<br/>(env: LLAMA_ARG_SPEC_DRAFT_BACKEND_SAMPLING) |
+| `--spec-draft-sampling {greedy,probabilistic}` | how the draft is sampled: greedy takes its argmax, probabilistic samples it and has the target verify by rejection sampling (default: greedy)<br/>(env: LLAMA_ARG_SPEC_DRAFT_SAMPLING) |
| `--spec-draft-device, -devd, --device-draft <dev1,dev2,..>` | comma-separated list of devices to use for offloading the draft model (none = don't offload, default: follows --device)<br/>use --list-devices to see a list of available devices |
| `--spec-draft-ngl, -ngld, --gpu-layers-draft, --n-gpu-layers-draft N` | max. number of draft model layers to store in VRAM, either an exact number, 'auto', or 'all' (default: auto)<br/>(env: LLAMA_ARG_N_GPU_LAYERS_DRAFT) |
| `--spec-draft-model, -md, --model-draft FNAME` | draft model for speculative decoding (default: unused)<br/>(env: LLAMA_ARG_SPEC_DRAFT_MODEL) |