Commit 185103dcf for llama.cpp
commit 185103dcf53222165ecd15ce8e406606e25bd091
Author: Aman Gupta <amangupta052@gmail.com>
Date: Wed Sep 30 20:27:18 2026 +0800
llama: llama_prefetch_rows (#29599)
* llama: llama_prefetch_rows
* llama: support row prefetch on Windows
Apply the Windows port contributed by @praneshgo unchanged.
Source: https://github.com/ggml-org/llama.cpp/pull/29599#issuecomment-5887721014
* avoid exposing llama-mmap in model code, route via llama-impl
* add windows check, only prefetch in lazy mode
* cont : clean-up
* cont : fix build
* cont : clarify padding token for gemma4
---------
Co-authored-by: Pranesh Gonegandla <pranesh.iitp@gmail.com>
Co-authored-by: Georgi Gerganov <ggerganov@gmail.com>
diff --git a/src/llama-impl.cpp b/src/llama-impl.cpp
index 5ec400e96..c216a310b 100644
--- a/src/llama-impl.cpp
+++ b/src/llama-impl.cpp
@@ -1,4 +1,5 @@
#include "llama-impl.h"
+#include "llama-mmap.h"
#include "ggml-backend.h"
#include "gguf.h"
@@ -19,6 +20,26 @@ struct llama_logger_state {
static llama_logger_state g_logger_state;
+void llama_prefetch_rows(const ggml_tensor * tensor, const int32_t * rows, size_t n_rows) {
+ if (!tensor || !tensor->data || !tensor->buffer || !ggml_backend_buffer_is_host(tensor->buffer) || n_rows == 0) {
+ return;
+ }
+
+ GGML_ASSERT(ggml_is_matrix(tensor));
+
+ const size_t row_bytes = ggml_row_size(tensor->type, tensor->ne[0]);
+ const auto * base = (const char *) tensor->data;
+
+ std::vector<llama_memory_range> mr;
+ mr.reserve(n_rows);
+ for (size_t i = 0; i < n_rows; ++i) {
+ GGML_ASSERT(rows[i] >= 0 && rows[i] < tensor->ne[1]);
+ mr.push_back({ base + (size_t) rows[i] * tensor->nb[1], row_bytes });
+ }
+
+ llama_prefetch(std::move(mr));
+}
+
time_meas::time_meas(int64_t & t_acc, bool disable) : t_start_us(disable ? -1 : ggml_time_us()), t_acc(t_acc) {}
time_meas::~time_meas() {
diff --git a/src/llama-impl.h b/src/llama-impl.h
index c34a6473b..7ccd23c86 100644
--- a/src/llama-impl.h
+++ b/src/llama-impl.h
@@ -74,6 +74,9 @@ static inline ggml_tensor * llama_mul_mat_hadamard(
return res;
}
+// Prefetch the host pages needed to gather these rows.
+void llama_prefetch_rows(const ggml_tensor * tensor, const int32_t * rows, size_t n_rows);
+
struct time_meas {
time_meas(int64_t & t_acc, bool disable = false);
~time_meas();
diff --git a/src/llama-mmap.cpp b/src/llama-mmap.cpp
index 715a6e354..6047c1061 100644
--- a/src/llama-mmap.cpp
+++ b/src/llama-mmap.cpp
@@ -821,6 +821,85 @@ const bool llama_mlock::SUPPORTED = true;
const bool llama_mlock::SUPPORTED = false;
#endif
+void llama_prefetch(llama_memory_ranges mr) {
+#if defined(__linux__) || (defined(_WIN32) && _WIN32_WINNT >= 0x602)
+ if (mr.empty()) {
+ return;
+ }
+
+#if defined(_WIN32)
+ using prefetch_virtual_memory_t = BOOL (WINAPI *)(HANDLE, ULONG_PTR, PWIN32_MEMORY_RANGE_ENTRY, ULONG);
+ static const auto pPrefetchVirtualMemory = (prefetch_virtual_memory_t) (void *) GetProcAddress(GetModuleHandleW(L"kernel32.dll"), "PrefetchVirtualMemory");
+ if (!pPrefetchVirtualMemory) {
+ return;
+ }
+
+ static const long page_size = [] {
+ SYSTEM_INFO info;
+ GetSystemInfo(&info);
+ return (long) info.dwPageSize;
+ }();
+#else
+ static const long page_size = sysconf(_SC_PAGESIZE);
+#endif
+ if (page_size <= 0) {
+ return;
+ }
+
+ const size_t page = (size_t) page_size;
+ std::sort(mr.begin(), mr.end(), [](const llama_memory_range & a, const llama_memory_range & b) {
+ return (uintptr_t) a.addr < (uintptr_t) b.addr;
+ });
+
+ uintptr_t begin = 0, end = 0;
+#if defined(_WIN32)
+ // collect the mr and prefetch them in one call, so the reads can be issued concurrently
+ std::vector<WIN32_MEMORY_RANGE_ENTRY> entries;
+ auto prefetch = [&]() {
+ entries.push_back({ (PVOID) begin, (SIZE_T) (end - begin) });
+ return true;
+ };
+#else
+ auto prefetch = [&]() {
+ if (madvise((void *) begin, end - begin, MADV_WILLNEED) != 0) {
+ LLAMA_LOG_WARN("llama_prefetch: madvise(MADV_WILLNEED) failed: %s\n", strerror(errno));
+ return false;
+ }
+ return true;
+ };
+#endif
+ for (const auto & range : mr) {
+ if (!range.addr || range.size == 0) {
+ continue;
+ }
+ const uintptr_t pointer = (uintptr_t) range.addr;
+ const uintptr_t first = pointer / page * page;
+ const uintptr_t last = (pointer + range.size + page - 1) / page * page;
+ if (end && first > end) {
+ if (!prefetch()) {
+ return;
+ }
+ end = 0;
+ }
+ if (!end) {
+ begin = first;
+ }
+ end = std::max(end, last);
+ }
+ if (end) {
+ prefetch();
+ }
+#if defined(_WIN32)
+ if (!entries.empty() && !pPrefetchVirtualMemory(GetCurrentProcess(), (ULONG_PTR) entries.size(), entries.data(), 0)) {
+ LLAMA_LOG_WARN("llama_prefetch: PrefetchVirtualMemory failed: %s\n",
+ llama_format_win_err(GetLastError()).c_str());
+ }
+#endif
+#else
+ GGML_UNUSED(mr);
+#endif
+}
+
size_t llama_path_max() {
return PATH_MAX;
}
diff --git a/src/llama-mmap.h b/src/llama-mmap.h
index cc28c8a73..e75945286 100644
--- a/src/llama-mmap.h
+++ b/src/llama-mmap.h
@@ -76,4 +76,14 @@ private:
std::unique_ptr<impl> pimpl;
};
+struct llama_memory_range {
+ const void * addr;
+ size_t size;
+};
+
+using llama_memory_ranges = std::vector<llama_memory_range>;
+
+// Prefetch the host pages covering these memory ranges.
+void llama_prefetch(llama_memory_ranges mr);
+
size_t llama_path_max();
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index 6ca536f22..eff60eea8 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -1784,8 +1784,15 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
}
}
}
+
ml.done_getting_tensors();
+ if (per_layer_tok_embd && ml.lazy.has(per_layer_tok_embd)) {
+ LLAMA_LOG_INFO("%s: enabling prefetch for '%s'\n", __func__, per_layer_tok_embd->name);
+
+ can_prefetch.insert(per_layer_tok_embd);
+ }
+
// Tied NVFP4 output is valid when no separate LM-head scale tensors are present.
// If sidecar scales exist, the output weight must be an actual output tensor.
GGML_ASSERT(!(output && tok_embd &&
diff --git a/src/llama-model.h b/src/llama-model.h
index 9c18ef045..8c4e438a2 100644
--- a/src/llama-model.h
+++ b/src/llama-model.h
@@ -734,6 +734,9 @@ struct llama_model {
// for keeping track of associated LoRA adapters
std::unordered_set<llama_adapter_lora *> loras;
+ // which tensors can be prefetched - driven by TENSOR_READ_LAZY
+ std::unordered_set<const ggml_tensor *> can_prefetch;
+
// statically allocated context for assigning
struct llama_meta_device_get_split_state_userdata get_split_state_ud;
diff --git a/src/models/gemma4.cpp b/src/models/gemma4.cpp
index 67de74c54..65fc7623d 100644
--- a/src/models/gemma4.cpp
+++ b/src/models/gemma4.cpp
@@ -1,4 +1,5 @@
#include "models.h"
+#include "llama-impl.h"
void llama_model_gemma4::load_arch_hparams(llama_model_loader & ml) {
hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
@@ -435,10 +436,40 @@ llama_model_gemma4::graph::graph(const llama_model & model, const llm_graph_para
ggml_build_forward_expand(gf, cur);
}
+class llm_graph_input_gemma4_ple : public llm_graph_input_i {
+public:
+ llm_graph_input_gemma4_ple(const llama_model & model) : model(model) {}
+
+ void set_input(const llama_ubatch * ubatch) override {
+ ggml_tensor * ple = model.per_layer_tok_embd;
+
+ const bool prefetch = model.can_prefetch.count(ple);
+
+ if (ubatch->token) {
+ if (prefetch) {
+ llama_prefetch_rows(ple, ubatch->token, ubatch->n_tokens);
+ }
+ ggml_backend_tensor_set(tokens, ubatch->token, 0, ubatch->n_tokens * ggml_element_size(tokens));
+ } else if (prefetch) {
+ // [TAG_GEMMA4_IMG_PADDING]
+ const int32_t padding = 0;
+ llama_prefetch_rows(ple, &padding, 1);
+ }
+ }
+
+ bool can_reuse(const llm_graph_params & params) override {
+ return params.ubatch.token ? tokens && tokens->ne[0] == params.ubatch.n_tokens : tokens == nullptr;
+ }
+
+ ggml_tensor * tokens = nullptr;
+
+ const llama_model & model;
+};
+
// equivalent to get_per_layer_inputs() in python code
// output shape: [n_embd_per_layer, n_layer, n_tokens]
ggml_tensor * llama_model_gemma4::graph::build_inp_per_layer() {
- auto inp = std::make_unique<llm_graph_input_embd>(n_embd);
+ auto inp = std::make_unique<llm_graph_input_gemma4_ple>(model);
ggml_tensor * inp_per_layer;
float tok_embd_scale = sqrtf((float) n_embd_per_layer);
@@ -451,9 +482,8 @@ ggml_tensor * llama_model_gemma4::graph::build_inp_per_layer() {
inp_per_layer = ggml_reshape_3d(ctx0, inp_per_layer, n_embd_per_layer, n_layer, n_tokens);
inp_per_layer = ggml_scale (ctx0, inp_per_layer, tok_embd_scale);
cb(inp_per_layer, "inp_per_layer_selected", -1);
-
- res->add_input(std::move(inp));
} else {
+ // [TAG_GEMMA4_IMG_PADDING]
// Multimodal embedding path: use padding token (ID=0) embedding
// TODO: verify if this is the correct behavior in transformers implementation
const int64_t embd_size = model.per_layer_tok_embd->ne[0]; // n_embd_per_layer * n_layer
@@ -467,6 +497,7 @@ ggml_tensor * llama_model_gemma4::graph::build_inp_per_layer() {
inp_per_layer = ggml_reshape_3d(ctx0, inp_per_layer, n_embd_per_layer, n_layer, 1);
cb(inp_per_layer, "inp_per_layer_multimodal", -1);
}
+ res->add_input(std::move(inp));
return inp_per_layer;
}
diff --git a/src/models/qwen4exp.cpp b/src/models/qwen4exp.cpp
index 319b7b9c1..13f3c19f3 100644
--- a/src/models/qwen4exp.cpp
+++ b/src/models/qwen4exp.cpp
@@ -1045,22 +1045,22 @@ ggml_tensor * llama_model_qwen4exp::graph::build_layer_ffn(ggml_tensor * cur, co
// mixed_n = (t[p]*m[0]) ^ ... ^ (t[p-n+1]*m[n-1]); row = mixed_n % vocab[h] + offset[h]
// The hash runs host-side because ggml has no int64 and no xor. EOS resets the window.
-class llm_graph_input_ple : public llm_graph_input_i {
+class llm_graph_input_qwen4exp_ple : public llm_graph_input_i {
public:
- llm_graph_input_ple(const llama_model_qwen4exp & pmodel,
- const llama_kv_cache_context * mctx) : pmodel(pmodel), mctx(mctx) {}
- virtual ~llm_graph_input_ple() = default;
+ llm_graph_input_qwen4exp_ple(const llama_model & model,
+ const llama_kv_cache_context * mctx) : model(model), mctx(mctx) {}
+ virtual ~llm_graph_input_qwen4exp_ple() = default;
void set_input(const llama_ubatch * ubatch) override;
bool can_reuse(const llm_graph_params & params) override {
mctx = static_cast<const llama_memory_hybrid_idx_context *>(params.mctx)->get_attn();
- return rows->ne[0] == (int64_t) pmodel.hparams.ple_n_heads * params.ubatch.n_tokens;
+ return rows->ne[0] == (int64_t) model.hparams.ple_n_heads * params.ubatch.n_tokens;
}
ggml_tensor * rows = nullptr; // I32 [ple_n_heads * n_tokens]
- const llama_model_qwen4exp & pmodel;
+ const llama_model & model;
// the predecessor tokens live in the attention KV cells (ext.tok)
const llama_kv_cache_context * mctx;
@@ -1069,24 +1069,24 @@ public:
std::vector<llama_token> prev;
};
-void llm_graph_input_ple::set_input(const llama_ubatch * ubatch) {
- const auto & hp = pmodel.hparams;
+void llm_graph_input_qwen4exp_ple::set_input(const llama_ubatch * ubatch) {
+ const auto & hparams = model.hparams;
// an image arrives as an embd batch, so ubatch->token is null, but every position still needs a row for ggml_get_rows
// stand in the image token id that the reference hashes, or EOS if the file has no such key
// gemma3n and gemma4 do the same with a hardcoded row 0 of per_layer_token_embd.
- const llama_token img_tok = hp.ple_image_token_id != 0
- ? (llama_token) hp.ple_image_token_id
- : (llama_token) hp.ple_eos_token_id;
+ const llama_token img_tok = hparams.ple_image_token_id != 0
+ ? (llama_token) hparams.ple_image_token_id
+ : (llama_token) hparams.ple_eos_token_id;
auto tok_of = [&](int64_t k) -> llama_token {
return ubatch->token ? ubatch->token[k] : img_tok;
};
const int64_t n_tokens = ubatch->n_tokens;
- const int64_t n_gram = hp.ple_ngram_size;
- const int64_t n_heads = hp.ple_n_heads;
- const int64_t per_gram = hp.ple_heads_per_ngram;
- const int64_t eos = hp.ple_eos_token_id;
+ const int64_t n_gram = hparams.ple_ngram_size;
+ const int64_t n_heads = hparams.ple_n_heads;
+ const int64_t per_gram = hparams.ple_heads_per_ngram;
+ const int64_t eos = hparams.ple_eos_token_id;
const int64_t n_prev = n_gram - 1;
std::vector<int32_t> idx(n_heads * n_tokens);
@@ -1116,19 +1116,28 @@ void llm_graph_input_ple::set_input(const llama_ubatch * ubatch) {
}
for (int64_t n = 2; n <= n_gram; ++n) {
- uint64_t mixed = (uint64_t) ctx[0] * hp.ple_layer_multipliers[0];
+ uint64_t mixed = (uint64_t) ctx[0] * hparams.ple_layer_multipliers[0];
for (int64_t j = 1; j < n; ++j) {
- mixed ^= (uint64_t) ctx[j] * hp.ple_layer_multipliers[j];
+ mixed ^= (uint64_t) ctx[j] * hparams.ple_layer_multipliers[j];
}
const int64_t base = (n - 2) * per_gram;
for (int64_t g = 0; g < per_gram; ++g) {
const int64_t h_i = base + g;
idx[i * n_heads + h_i] =
- (int32_t) (mixed % hp.ple_head_vocab_sizes[h_i] + hp.ple_head_offsets[h_i]);
+ (int32_t) (mixed % hparams.ple_head_vocab_sizes[h_i] + hparams.ple_head_offsets[h_i]);
}
}
}
+ {
+ ggml_tensor * ple = model.per_layer_tok_embd;
+
+ const bool prefetch = model.can_prefetch.count(ple);
+ if (prefetch) {
+ llama_prefetch_rows(ple, idx.data(), idx.size());
+ }
+ }
+
ggml_backend_tensor_set(rows, idx.data(), 0, idx.size()*ggml_element_size(rows));
}
@@ -1193,8 +1202,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_inp_ple(
const int64_t n_heads = hparams.ple_n_heads;
// the attention cells see every ubatch regardless of the layer types
- auto ple_inp = std::make_unique<llm_graph_input_ple>(
- static_cast<const llama_model_qwen4exp &>(model), mctx_hyb->get_attn());
+ auto ple_inp = std::make_unique<llm_graph_input_qwen4exp_ple>(model, mctx_hyb->get_attn());
ple_inp->rows = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_heads * n_tokens);
ggml_set_input(ple_inp->rows);