Commit e60eff95f for llama.cpp
commit e60eff95fda7b6e46bf402a8e9f1e86d88ab29c8
Author: Georgi Gerganov <ggerganov@gmail.com>
Date: Fri Oct 9 16:01:04 2026 +0300
meta : handle host views (#30217)
* meta : handle views of tensors allocated on the host
A view shares the memory of its view_src, so ggml-alloc never allocates a view in
the buffer of the split it lands in - the scheduler copies the source into the
split and the ops that use the view read that copy. The view node itself is a noop
and does not need a split of its own, but the meta backend asserted when one was
left inside a meta split:
- ggml_backend_meta_get_split_state() dereferenced tensor->buffer->context
- the graph rebuild mapped every node with ggml_backend_meta_buffer_simple_tensor()
Accept such nodes when they are views of host tensors, which also generalizes the
previous s_copy_main workaround. This fixes the assert hit by KV cache views when
using --split-mode tensor with partial offload.
Assisted-by: pi:llama.cpp/Qwen3.8-Flash-Next
* archs : re-enable sm tensor for K2 Horizon
* cont : add TODO and reference
diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp
index 8ed5f4ebb..86748b689 100644
--- a/ggml/src/ggml-backend-meta.cpp
+++ b/ggml/src/ggml-backend-meta.cpp
@@ -1176,7 +1176,22 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(
return ret;
}
+static bool ggml_backend_meta_is_host_view(const struct ggml_tensor * tensor) {
+ return ggml_is_view(tensor) && ggml_backend_buffer_is_host(tensor->view_src->buffer);
+}
+
static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(const struct ggml_tensor * tensor, bool assume_sync) {
+ // [TAG_META_HOST_VIEWS]
+ // TODO: technically, this check should not be needed if the backend scheduler correctly prevents assigning
+ // such host-buffer views to the meta backend. figure out how to update the scheduler logic to achieve that
+ // ref: https://github.com/ggml-org/llama.cpp/pull/30217
+ if (!ggml_backend_buffer_is_meta(tensor->buffer)) {
+ GGML_ASSERT(ggml_backend_meta_is_host_view(tensor));
+
+ // the view is not allocated in the meta buffer, it is not split across the sub-devices
+ return { GGML_BACKEND_SPLIT_AXIS_NONE, {0}, {1}, 1 };
+ }
+
ggml_backend_meta_buffer_context * buf_ctx = (ggml_backend_meta_buffer_context *) tensor->buffer->context;
return ggml_backend_meta_get_split_state(buf_ctx->get_simple_tensor_container(tensor), tensor, assume_sync);
}
@@ -2026,9 +2041,11 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend,
for (int i = 0; i < cgraph->n_nodes; i++) {
ggml_tensor * node = cgraph->nodes[i];
- if (node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && ggml_backend_buffer_is_host(node->view_src->buffer)) {
- // FIXME s_copy_main is on the CPU and its view seems to be incorrectly added to the graph nodes.
- // For regular usage this doesn't matter since it's a noop but trying to call ggml_backend_meta_buffer_simple_tensor results in a crash.
+ if (!ggml_backend_buffer_is_meta(node->buffer)) {
+ // [TAG_META_HOST_VIEWS]
+ GGML_ASSERT(ggml_backend_meta_is_host_view(node));
+
+ // keep the node as is, mapping it to a simple tensor is not possible
bcj.nodes[i] = node;
continue;
}
diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
index 821479064..5774867d9 100644
--- a/src/llama-arch.cpp
+++ b/src/llama-arch.cpp
@@ -1234,7 +1234,6 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {
case LLM_ARCH_KIMI_K3:
case LLM_ARCH_GLM5_NEXT:
case LLM_ARCH_QWEN3TTS:
- case LLM_ARCH_K2_HORIZON:
return false;
default:
return true;