Commit 760984655 for llama.cpp
commit 7609846557c50f9d984719a9e1e8c5f3d02f807b
Author: Patrick Hoffmann <patrick@hoffmann.lol>
Date: Tue Sep 15 13:50:20 2026 +0200
rpc : hash-cache only weights (#28789)
* rpc : hash-cache only weights
ggml_backend_rpc_buffer_set_tensor and ggml_backend_rpc_set_tensor_async
hashed every transfer above HASH_THRESHOLD and let `rpc-server -c` serve it
from its file cache. The cache is meant for weights, but the activations
ggml_backend_sched copies between backends took the same path: with a
two-node split of Qwen3.8-Flash-Next every prefill ubatch above 10 MB was
hashed, written to the worker's cache directory (1.4 TB after a day) and
later served from there. Use the hash path only for tensors in buffers
marked GGML_BACKEND_BUFFER_USAGE_WEIGHTS.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
* rpc : save a cache entry only for the tensor that missed the hash check
With the client hashing weights only, the server still wrote every
SET_TENSOR above HASH_THRESHOLD to the cache directory, so the compute
data the scheduler sends kept filling the disk. Remember the hash of the
last SET_TENSOR_HASH that missed and save only the SET_TENSOR that
follows it with that hash - the weight the client is re-sending.
* rpc : signal the cache decision in the SET_TENSOR payload
Replace the server-side `pending_cache` state with a `cache_flag` byte
in the SET_TENSOR message: the client sets it when SET_TENSOR_HASH
reported a miss, the server saves a cache entry only when it is set.
Bump RPC_PROTO_MAJOR_VERSION since the wire format changes.
---------
Co-authored-by: Patrick Hoffmann <patrickhoffmann@MacBook-Pro-14-HOP.local>
Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
diff --git a/ggml/include/ggml-rpc.h b/ggml/include/ggml-rpc.h
index cbfe40013..1f8cb7906 100644
--- a/ggml/include/ggml-rpc.h
+++ b/ggml/include/ggml-rpc.h
@@ -6,7 +6,7 @@
extern "C" {
#endif
-#define RPC_PROTO_MAJOR_VERSION 6
+#define RPC_PROTO_MAJOR_VERSION 7
#define RPC_PROTO_MINOR_VERSION 0
#define RPC_PROTO_PATCH_VERSION 0
diff --git a/ggml/src/ggml-rpc/ggml-rpc.cpp b/ggml/src/ggml-rpc/ggml-rpc.cpp
index cc7d72069..adb88a245 100644
--- a/ggml/src/ggml-rpc/ggml-rpc.cpp
+++ b/ggml/src/ggml-rpc/ggml-rpc.cpp
@@ -697,10 +697,31 @@ static void ggml_backend_rpc_buffer_memset_tensor(
ctx->dispatcher->send(RPC_CMD_MEMSET_TENSOR, request, sizeof(*request));
}
+// input serialization format: | rpc_tensor | cache_flag (1 byte) | offset (8 bytes) | data (size bytes)
+static std::shared_ptr<uint8_t> serialize_set_tensor(const rpc_tensor & rpc_tensor, uint8_t cache_flag, uint64_t offset, const void * data, size_t size, size_t & input_size) {
+ input_size = sizeof(rpc_tensor) + sizeof(cache_flag) + sizeof(offset) + size;
+ uint8_t * input = new uint8_t[input_size]();
+ uint8_t * p = input;
+ memcpy(p, &rpc_tensor, sizeof(rpc_tensor)); p += sizeof(rpc_tensor);
+ memcpy(p, &cache_flag, sizeof(cache_flag)); p += sizeof(cache_flag);
+ memcpy(p, &offset, sizeof(offset)); p += sizeof(offset);
+ memcpy(p, data, size);
+ return std::shared_ptr<uint8_t>(input, std::default_delete<uint8_t[]>());
+}
+
+// the hash cache is meant for weights, so that a model reload can skip re-sending them.
+// compute-buffer inputs (the activations ggml_backend_sched copies between backends) must not
+// take this path, otherwise with `rpc-server -c` every ubatch above the threshold is written
+// to the cache directory and later served from there.
+static bool rpc_use_hash_cache(const ggml_tensor * tensor, size_t size) {
+ return size > HASH_THRESHOLD && tensor->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS;
+}
+
static void ggml_backend_rpc_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) {
ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context;
rpc_tensor rpc_tensor = serialize_tensor(tensor);
- if (size > HASH_THRESHOLD) {
+ uint8_t cache_flag = 0;
+ if (rpc_use_hash_cache(tensor, size)) {
auto request = std::make_shared<rpc_msg_set_tensor_hash_req>();
request->tensor = rpc_tensor;
request->offset = offset;
@@ -711,15 +732,12 @@ static void ggml_backend_rpc_buffer_set_tensor(ggml_backend_buffer_t buffer, ggm
// the server has the same data, no need to send it
return;
}
+ // the server has no cache entry for this tensor - ask it to save one
+ cache_flag = 1;
}
- // input serialization format: | rpc_tensor | offset (8 bytes) | data (size bytes)
- size_t input_size = sizeof(rpc_tensor) + sizeof(uint64_t) + size;
- uint8_t * input = new uint8_t[input_size]();
- memcpy(input, &rpc_tensor, sizeof(rpc_tensor));
- memcpy(input + sizeof(rpc_tensor), &offset, sizeof(offset));
- memcpy(input + sizeof(rpc_tensor) + sizeof(offset), data, size);
- std::shared_ptr<uint8_t> input_ptr(input, std::default_delete<uint8_t[]>());
- ctx->dispatcher->send(RPC_CMD_SET_TENSOR, input_ptr, input_size);
+ size_t input_size;
+ auto input = serialize_set_tensor(rpc_tensor, cache_flag, offset, data, size, input_size);
+ ctx->dispatcher->send(RPC_CMD_SET_TENSOR, input, input_size);
}
static void ggml_backend_rpc_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size) {
@@ -927,7 +945,8 @@ static void ggml_backend_rpc_free(ggml_backend_t backend) {
static void ggml_backend_rpc_set_tensor_async(ggml_backend_t backend, ggml_tensor * tensor, const void * data, size_t offset, size_t size) {
ggml_backend_rpc_context * ctx = (ggml_backend_rpc_context *)backend->context;
rpc_tensor rpc_tensor = serialize_tensor(tensor);
- if (size > HASH_THRESHOLD) {
+ uint8_t cache_flag = 0;
+ if (rpc_use_hash_cache(tensor, size)) {
auto request = std::make_shared<rpc_msg_set_tensor_hash_req>();
request->tensor = rpc_tensor;
request->offset = offset;
@@ -939,15 +958,12 @@ static void ggml_backend_rpc_set_tensor_async(ggml_backend_t backend, ggml_tenso
// the server has the same data, no need to send it
return;
}
+ // the server has no cache entry for this tensor - ask it to save one
+ cache_flag = 1;
}
- // input serialization format: | rpc_tensor | offset (8 bytes) | data (size bytes)
- size_t input_size = sizeof(rpc_tensor) + sizeof(uint64_t) + size;
- uint8_t * input = new uint8_t[input_size]();
- memcpy(input, &rpc_tensor, sizeof(rpc_tensor));
- memcpy(input + sizeof(rpc_tensor), &offset, sizeof(offset));
- memcpy(input + sizeof(rpc_tensor) + sizeof(offset), data, size);
- std::shared_ptr<uint8_t> input_ptr(input, std::default_delete<uint8_t[]>());
- ctx->dispatcher->send_async(RPC_CMD_SET_TENSOR, input_ptr, input_size);
+ size_t input_size;
+ auto input = serialize_set_tensor(rpc_tensor, cache_flag, offset, data, size, input_size);
+ ctx->dispatcher->send_async(RPC_CMD_SET_TENSOR, input, input_size);
}
static void ggml_backend_rpc_get_tensor_async(ggml_backend_t backend, const ggml_tensor * tensor, void * data, size_t offset, size_t size) {
@@ -1398,14 +1414,17 @@ ggml_tensor * rpc_server::deserialize_tensor(struct ggml_context * ctx, const rp
bool rpc_server::set_tensor(const std::vector<uint8_t> & input) {
- // serialization format: | rpc_tensor | offset (8 bytes) | data (size bytes) |
- if (input.size() < sizeof(rpc_tensor) + sizeof(uint64_t)) {
+ // serialization format: | rpc_tensor | cache_flag (1 byte) | offset (8 bytes) | data (size bytes) |
+ uint8_t cache_flag;
+ uint64_t offset;
+ const size_t header_size = sizeof(rpc_tensor) + sizeof(cache_flag) + sizeof(offset);
+ if (input.size() < header_size) {
return false;
}
const rpc_tensor * in_tensor = (const rpc_tensor *)input.data();
- uint64_t offset;
- memcpy(&offset, input.data() + sizeof(rpc_tensor), sizeof(offset));
- const size_t size = input.size() - sizeof(rpc_tensor) - sizeof(offset);
+ memcpy(&cache_flag, input.data() + sizeof(rpc_tensor), sizeof(cache_flag));
+ memcpy(&offset, input.data() + sizeof(rpc_tensor) + sizeof(cache_flag), sizeof(offset));
+ const size_t size = input.size() - header_size;
struct ggml_init_params params {
/*.mem_size =*/ ggml_tensor_overhead(),
@@ -1434,8 +1453,8 @@ bool rpc_server::set_tensor(const std::vector<uint8_t> & input) {
}
}
- const void * data = input.data() + sizeof(rpc_tensor) + sizeof(offset);
- if (cache_dir && size > HASH_THRESHOLD) {
+ const void * data = input.data() + header_size;
+ if (cache_dir && cache_flag) {
uint64_t hash = fnv_hash((const uint8_t*)data, size);
char hash_str[17];
snprintf(hash_str, sizeof(hash_str), "%016" PRIx64, hash);