From 9f80c69e3b03855a4e286e48755613438b19b5ce Mon Sep 17 00:00:00 2001 From: Aman Gupta Date: Mon, 5 Oct 2026 17:50:17 +0800 Subject: [PATCH] CUDA: make the alloc_deps check batch independent Fixes #29980 --- ggml/src/ggml-cuda/ggml-cuda.cu | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu index c72419613d..e068116f3e 100644 --- a/ggml/src/ggml-cuda/ggml-cuda.cu +++ b/ggml/src/ggml-cuda/ggml-cuda.cu @@ -1846,11 +1846,14 @@ static bool ggml_cuda_match_shared_expert(const ggml_cgraph * graph, int routed_ return (a == graph->nodes[idx] && b == graph->nodes[idx + 1]) || (b == graph->nodes[idx] && a == graph->nodes[idx + 1]); }; + // only batch-size independent checks here: graph_optimize must produce the same graph topology for every ubatch + // size, otherwise ggml-alloc has to re-reserve (and the scheduler to synchronize) at runtime. + // the MMVQ batch size check is done in ggml_cuda_try_fuse if (!is_pair(gate, up, routed_idx) || !is_pair(shared_gate, shared_up, shared_idx) || !ggml_cuda_should_fuse_mul_mat(up, gate, routed) || !ggml_cuda_should_fuse_mul_mat(shared_up, shared_gate, shared) || !up->src[0]->buffer || - !ggml_cuda_should_fuse_mul_mat_vec_q(up)) { + !ggml_is_quantized(up->src[0]->type)) { return false; } const ggml_tensor * input = up->src[1]; @@ -3510,7 +3513,8 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph ggml_tensor * node = cgraph->nodes[i]; if (node->op == GGML_OP_MUL_MAT_ID && cuda_ctx->stream_context().concurrent_events.empty() && - ggml_cuda_match_shared_expert(cgraph, i, i + 3)) { + ggml_cuda_match_shared_expert(cgraph, i, i + 3) && + ggml_cuda_should_fuse_mul_mat_vec_q(cgraph->nodes[i + 2]->src[1])) { const int outputs[] = { i + 2, i + 5 }; if (ggml_cuda_check_fusion_memory_ranges(cgraph, i, 6, outputs, 2)) { ggml_tensor * routed = cgraph->nodes[i + 2];