mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-11 07:20:33 +02:00
ggml : collect all input tensors into graph_inputs (#29634)
graph_inputs was populated while splitting the graph, so it only contained the inputs that are used as srcs of some node. With pipeline parallelism (n_copies > 1), each graph input contributes n_copies leafs to graph_copy, so switching between batches that consume different inputs (e.g. token batches that do not use the embeddings input vs image batches that do) changed the graph composition. This shifted the input copies in graph_copy, making the backend ids comparison report spurious changes and forcing the scheduler to re-reserve. The re-reserve could then record smaller input sizes (e.g. out_ids with n_outputs = 0) and abort later on a graph with an unchanged size via GGML_SCHED_DEBUG_REALLOC. Collect the inputs after the split instead, from all input leafs of the graph, so that the graph composition depends only on which inputs exist, not on which inputs are used. Assisted-by: pi:llama.cpp/MiMo-V2.6-Flash-RL
This commit is contained in:
+40
-24
@@ -1372,30 +1372,6 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra
|
||||
const int src_backend_id = sched->hv_tensor_backend_ids[src_id];
|
||||
GGML_ASSERT(src_backend_id != -1); // all inputs should be assigned by now
|
||||
|
||||
if (src->flags & GGML_TENSOR_FLAG_INPUT && sched->n_copies > 1) {
|
||||
if (tensor_id_copy(src_id, src_backend_id, 0) == NULL) {
|
||||
ggml_backend_t backend = sched->backends[src_backend_id];
|
||||
for (int c = 0; c < sched->n_copies; c++) {
|
||||
struct ggml_tensor * tensor_copy;
|
||||
if (c == sched->cur_copy) {
|
||||
tensor_copy = src; // use the original tensor as the current copy
|
||||
} else {
|
||||
tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);
|
||||
ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);
|
||||
}
|
||||
ggml_set_input(tensor_copy);
|
||||
ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor
|
||||
tensor_id_copy(src_id, src_backend_id, c) = tensor_copy;
|
||||
SET_CAUSE(tensor_copy, "4.cpy");
|
||||
}
|
||||
int n_graph_inputs = sched->n_graph_inputs++;
|
||||
if (n_graph_inputs >= sched->graph_inputs_capacity) {
|
||||
ggml_backend_sched_graph_inputs_grow(sched);
|
||||
}
|
||||
sched->graph_inputs[n_graph_inputs] = src;
|
||||
}
|
||||
}
|
||||
|
||||
if (src_backend_id != cur_backend_id && !ggml_backend_sched_buffer_supported(sched, src, cur_backend_id)) {
|
||||
// create a copy of the input in the split's backend
|
||||
if (tensor_id_copy(src_id, cur_backend_id, 0) == NULL) {
|
||||
@@ -1428,6 +1404,46 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra
|
||||
ggml_backend_sched_print_assignments(sched, graph);
|
||||
}
|
||||
|
||||
// pass 6: collect all input tensors into graph_inputs
|
||||
// this includes inputs not consumed by any node (e.g. the embeddings input of a text-only batch) so that
|
||||
// the graph composition does not depend on which inputs are used, which would otherwise cause graph
|
||||
// reallocations when switching between different types of batches [GGML_SCHED_DEBUG_REALLOC]
|
||||
if (sched->n_copies > 1) {
|
||||
for (int i = 0; i < graph->n_leafs; i++) {
|
||||
struct ggml_tensor * leaf = graph->leafs[i];
|
||||
if ((leaf->flags & GGML_TENSOR_FLAG_INPUT) == 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const size_t leaf_id = hash_id(leaf);
|
||||
const int leaf_backend_id = tensor_backend_id(leaf);
|
||||
GGML_ASSERT(leaf_backend_id != -1); // all leafs should be assigned by now
|
||||
|
||||
if (tensor_id_copy(leaf_id, leaf_backend_id, 0) == NULL) {
|
||||
ggml_backend_t backend = sched->backends[leaf_backend_id];
|
||||
for (int c = 0; c < sched->n_copies; c++) {
|
||||
struct ggml_tensor * tensor_copy;
|
||||
if (c == sched->cur_copy) {
|
||||
tensor_copy = leaf; // use the original tensor as the current copy
|
||||
} else {
|
||||
tensor_copy = ggml_dup_tensor_layout(sched->ctx, leaf);
|
||||
ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), leaf->name, c);
|
||||
}
|
||||
ggml_set_input(tensor_copy);
|
||||
ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor
|
||||
tensor_id_copy(leaf_id, leaf_backend_id, c) = tensor_copy;
|
||||
SET_CAUSE(tensor_copy, "6.cpy");
|
||||
}
|
||||
}
|
||||
|
||||
int n_graph_inputs = sched->n_graph_inputs++;
|
||||
if (n_graph_inputs >= sched->graph_inputs_capacity) {
|
||||
ggml_backend_sched_graph_inputs_grow(sched);
|
||||
}
|
||||
sched->graph_inputs[n_graph_inputs] = leaf;
|
||||
}
|
||||
}
|
||||
|
||||
// swap node_backend_ids and leaf _backend_ids with prevs
|
||||
{
|
||||
int * tmp = sched->node_backend_ids;
|
||||
|
||||
Reference in New Issue
Block a user