ggml : collect all input tensors into graph_inputs (#29634)

graph_inputs was populated while splitting the graph, so it only
contained the inputs that are used as srcs of some node. With pipeline
parallelism (n_copies > 1), each graph input contributes n_copies leafs
to graph_copy, so switching between batches that consume different
inputs (e.g. token batches that do not use the embeddings input vs
image batches that do) changed the graph composition. This shifted the
input copies in graph_copy, making the backend ids comparison report
spurious changes and forcing the scheduler to re-reserve. The
re-reserve could then record smaller input sizes (e.g. out_ids with
n_outputs = 0) and abort later on a graph with an unchanged size via
GGML_SCHED_DEBUG_REALLOC.

Collect the inputs after the split instead, from all input leafs of the
graph, so that the graph composition depends only on which inputs
exist, not on which inputs are used.

Assisted-by: pi:llama.cpp/MiMo-V2.6-Flash-RL
This commit is contained in:
Georgi Gerganov
2026-09-29 13:37:50 +03:00
committed by GitHub
parent 86ea01d05e
commit 8019dc563b
+40 -24
View File
@@ -1372,30 +1372,6 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra
const int src_backend_id = sched->hv_tensor_backend_ids[src_id];
GGML_ASSERT(src_backend_id != -1); // all inputs should be assigned by now
if (src->flags & GGML_TENSOR_FLAG_INPUT && sched->n_copies > 1) {
if (tensor_id_copy(src_id, src_backend_id, 0) == NULL) {
ggml_backend_t backend = sched->backends[src_backend_id];
for (int c = 0; c < sched->n_copies; c++) {
struct ggml_tensor * tensor_copy;
if (c == sched->cur_copy) {
tensor_copy = src; // use the original tensor as the current copy
} else {
tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);
ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);
}
ggml_set_input(tensor_copy);
ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor
tensor_id_copy(src_id, src_backend_id, c) = tensor_copy;
SET_CAUSE(tensor_copy, "4.cpy");
}
int n_graph_inputs = sched->n_graph_inputs++;
if (n_graph_inputs >= sched->graph_inputs_capacity) {
ggml_backend_sched_graph_inputs_grow(sched);
}
sched->graph_inputs[n_graph_inputs] = src;
}
}
if (src_backend_id != cur_backend_id && !ggml_backend_sched_buffer_supported(sched, src, cur_backend_id)) {
// create a copy of the input in the split's backend
if (tensor_id_copy(src_id, cur_backend_id, 0) == NULL) {
@@ -1428,6 +1404,46 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra
ggml_backend_sched_print_assignments(sched, graph);
}
// pass 6: collect all input tensors into graph_inputs
// this includes inputs not consumed by any node (e.g. the embeddings input of a text-only batch) so that
// the graph composition does not depend on which inputs are used, which would otherwise cause graph
// reallocations when switching between different types of batches [GGML_SCHED_DEBUG_REALLOC]
if (sched->n_copies > 1) {
for (int i = 0; i < graph->n_leafs; i++) {
struct ggml_tensor * leaf = graph->leafs[i];
if ((leaf->flags & GGML_TENSOR_FLAG_INPUT) == 0) {
continue;
}
const size_t leaf_id = hash_id(leaf);
const int leaf_backend_id = tensor_backend_id(leaf);
GGML_ASSERT(leaf_backend_id != -1); // all leafs should be assigned by now
if (tensor_id_copy(leaf_id, leaf_backend_id, 0) == NULL) {
ggml_backend_t backend = sched->backends[leaf_backend_id];
for (int c = 0; c < sched->n_copies; c++) {
struct ggml_tensor * tensor_copy;
if (c == sched->cur_copy) {
tensor_copy = leaf; // use the original tensor as the current copy
} else {
tensor_copy = ggml_dup_tensor_layout(sched->ctx, leaf);
ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), leaf->name, c);
}
ggml_set_input(tensor_copy);
ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor
tensor_id_copy(leaf_id, leaf_backend_id, c) = tensor_copy;
SET_CAUSE(tensor_copy, "6.cpy");
}
}
int n_graph_inputs = sched->n_graph_inputs++;
if (n_graph_inputs >= sched->graph_inputs_capacity) {
ggml_backend_sched_graph_inputs_grow(sched);
}
sched->graph_inputs[n_graph_inputs] = leaf;
}
}
// swap node_backend_ids and leaf _backend_ids with prevs
{
int * tmp = sched->node_backend_ids;