Commit 8019dc563 for llama.cpp
commit 8019dc563b1ecbae6b161a70c3a1359f1b206c1e
Author: Georgi Gerganov <ggerganov@gmail.com>
Date: Tue Sep 29 13:37:50 2026 +0300
ggml : collect all input tensors into graph_inputs (#29634)
graph_inputs was populated while splitting the graph, so it only
contained the inputs that are used as srcs of some node. With pipeline
parallelism (n_copies > 1), each graph input contributes n_copies leafs
to graph_copy, so switching between batches that consume different
inputs (e.g. token batches that do not use the embeddings input vs
image batches that do) changed the graph composition. This shifted the
input copies in graph_copy, making the backend ids comparison report
spurious changes and forcing the scheduler to re-reserve. The
re-reserve could then record smaller input sizes (e.g. out_ids with
n_outputs = 0) and abort later on a graph with an unchanged size via
GGML_SCHED_DEBUG_REALLOC.
Collect the inputs after the split instead, from all input leafs of the
graph, so that the graph composition depends only on which inputs
exist, not on which inputs are used.
Assisted-by: pi:llama.cpp/MiMo-V2.6-Flash-RL
diff --git a/ggml/src/ggml-backend.cpp b/ggml/src/ggml-backend.cpp
index 20bf96501..273dc92b2 100644
--- a/ggml/src/ggml-backend.cpp
+++ b/ggml/src/ggml-backend.cpp
@@ -1372,30 +1372,6 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra
const int src_backend_id = sched->hv_tensor_backend_ids[src_id];
GGML_ASSERT(src_backend_id != -1); // all inputs should be assigned by now
- if (src->flags & GGML_TENSOR_FLAG_INPUT && sched->n_copies > 1) {
- if (tensor_id_copy(src_id, src_backend_id, 0) == NULL) {
- ggml_backend_t backend = sched->backends[src_backend_id];
- for (int c = 0; c < sched->n_copies; c++) {
- struct ggml_tensor * tensor_copy;
- if (c == sched->cur_copy) {
- tensor_copy = src; // use the original tensor as the current copy
- } else {
- tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);
- ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);
- }
- ggml_set_input(tensor_copy);
- ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor
- tensor_id_copy(src_id, src_backend_id, c) = tensor_copy;
- SET_CAUSE(tensor_copy, "4.cpy");
- }
- int n_graph_inputs = sched->n_graph_inputs++;
- if (n_graph_inputs >= sched->graph_inputs_capacity) {
- ggml_backend_sched_graph_inputs_grow(sched);
- }
- sched->graph_inputs[n_graph_inputs] = src;
- }
- }
-
if (src_backend_id != cur_backend_id && !ggml_backend_sched_buffer_supported(sched, src, cur_backend_id)) {
// create a copy of the input in the split's backend
if (tensor_id_copy(src_id, cur_backend_id, 0) == NULL) {
@@ -1428,6 +1404,46 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra
ggml_backend_sched_print_assignments(sched, graph);
}
+ // pass 6: collect all input tensors into graph_inputs
+ // this includes inputs not consumed by any node (e.g. the embeddings input of a text-only batch) so that
+ // the graph composition does not depend on which inputs are used, which would otherwise cause graph
+ // reallocations when switching between different types of batches [GGML_SCHED_DEBUG_REALLOC]
+ if (sched->n_copies > 1) {
+ for (int i = 0; i < graph->n_leafs; i++) {
+ struct ggml_tensor * leaf = graph->leafs[i];
+ if ((leaf->flags & GGML_TENSOR_FLAG_INPUT) == 0) {
+ continue;
+ }
+
+ const size_t leaf_id = hash_id(leaf);
+ const int leaf_backend_id = tensor_backend_id(leaf);
+ GGML_ASSERT(leaf_backend_id != -1); // all leafs should be assigned by now
+
+ if (tensor_id_copy(leaf_id, leaf_backend_id, 0) == NULL) {
+ ggml_backend_t backend = sched->backends[leaf_backend_id];
+ for (int c = 0; c < sched->n_copies; c++) {
+ struct ggml_tensor * tensor_copy;
+ if (c == sched->cur_copy) {
+ tensor_copy = leaf; // use the original tensor as the current copy
+ } else {
+ tensor_copy = ggml_dup_tensor_layout(sched->ctx, leaf);
+ ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), leaf->name, c);
+ }
+ ggml_set_input(tensor_copy);
+ ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor
+ tensor_id_copy(leaf_id, leaf_backend_id, c) = tensor_copy;
+ SET_CAUSE(tensor_copy, "6.cpy");
+ }
+ }
+
+ int n_graph_inputs = sched->n_graph_inputs++;
+ if (n_graph_inputs >= sched->graph_inputs_capacity) {
+ ggml_backend_sched_graph_inputs_grow(sched);
+ }
+ sched->graph_inputs[n_graph_inputs] = leaf;
+ }
+ }
+
// swap node_backend_ids and leaf _backend_ids with prevs
{
int * tmp = sched->node_backend_ids;