Commit 32dd62ee6 for llama.cpp
commit 32dd62ee6dfa80ada846551fefec215cefc5ae1c
Author: Pranesh Gonegandla <pranesh.iitp@gmail.com>
Date: Thu Oct 1 13:14:31 2026 +0530
llama-mmap : avoid a second full-size copy of each tensor with direct-io (#29749)
Assisted-by: Claude
Co-authored-by: Pranesh Gonegandla <pgonegandla@nvidia.com>
diff --git a/src/llama-mmap.cpp b/src/llama-mmap.cpp
index 6047c1061..045489df7 100644
--- a/src/llama-mmap.cpp
+++ b/src/llama-mmap.cpp
@@ -323,8 +323,11 @@ struct llama_file::impl {
off_t offset_from_alignment = offset - aligned_offset;
size_t bytes_to_read = (offset_from_alignment + size + alignment - 1) & ~(alignment - 1);
+ // stage through a bounded buffer, so that a large tensor is not held in memory twice while it loads
+ const size_t buffer_size = std::min<size_t>(bytes_to_read, LLAMA_DIRECT_IO_BUFFER_SIZE);
+
void * raw_buffer = nullptr;
- int ret = posix_memalign(&raw_buffer, alignment, bytes_to_read);
+ int ret = posix_memalign(&raw_buffer, alignment, buffer_size);
if (ret != 0) {
throw std::runtime_error(format("posix_memalign failed with error %d", ret));
}
@@ -335,10 +338,20 @@ struct llama_file::impl {
std::unique_ptr<void, aligned_buffer_deleter> buffer(raw_buffer);
seek(aligned_offset, SEEK_SET);
- read_raw_unsafe(buffer.get(), bytes_to_read);
- uintptr_t actual_data = reinterpret_cast<uintptr_t>(buffer.get()) + offset_from_alignment;
- memcpy(dest, reinterpret_cast<void *>(actual_data), size);
+ size_t skip = offset_from_alignment;
+ size_t copied = 0;
+ for (size_t done = 0; done < bytes_to_read; ) {
+ const size_t n = std::min(buffer_size, bytes_to_read - done);
+ read_raw_unsafe(buffer.get(), n);
+
+ const size_t count = std::min(n - skip, size - copied);
+ memcpy(reinterpret_cast<char *>(dest) + copied, reinterpret_cast<char *>(buffer.get()) + skip, count);
+
+ copied += count;
+ skip = 0;
+ done += n;
+ }
}
void read_raw(void * ptr, size_t len) {
diff --git a/src/llama-mmap.h b/src/llama-mmap.h
index e75945286..6c6393d57 100644
--- a/src/llama-mmap.h
+++ b/src/llama-mmap.h
@@ -6,6 +6,9 @@
#include <vector>
#include <cstdio>
+// staging buffer size for direct I/O reads, 64MB works well for NVMe drives
+#define LLAMA_DIRECT_IO_BUFFER_SIZE (64 * 1024 * 1024)
+
struct llama_file;
struct llama_mmap;
struct llama_mlock;
diff --git a/src/llama-model-loader.cpp b/src/llama-model-loader.cpp
index 43c396f15..6cc0b2028 100644
--- a/src/llama-model-loader.cpp
+++ b/src/llama-model-loader.cpp
@@ -1516,8 +1516,7 @@ bool llama_model_loader::load_all_data(
}
// Buffer size: balance between memory usage and I/O efficiency
- // 64MB works well for NVMe drives
- const size_t buffer_size = alignment != 1 ? 64 * 1024 * 1024 + 2 * alignment : 1 * 1024 * 1024;
+ const size_t buffer_size = alignment != 1 ? LLAMA_DIRECT_IO_BUFFER_SIZE + 2 * alignment : 1 * 1024 * 1024;
std::vector<ggml_backend_buffer_t> host_buffers;
std::vector<ggml_backend_event_t> events;