mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-08 22:10:37 +02:00
@@ -1846,11 +1846,14 @@ static bool ggml_cuda_match_shared_expert(const ggml_cgraph * graph, int routed_
|
|||||||
return (a == graph->nodes[idx] && b == graph->nodes[idx + 1]) ||
|
return (a == graph->nodes[idx] && b == graph->nodes[idx + 1]) ||
|
||||||
(b == graph->nodes[idx] && a == graph->nodes[idx + 1]);
|
(b == graph->nodes[idx] && a == graph->nodes[idx + 1]);
|
||||||
};
|
};
|
||||||
|
// only batch-size independent checks here: graph_optimize must produce the same graph topology for every ubatch
|
||||||
|
// size, otherwise ggml-alloc has to re-reserve (and the scheduler to synchronize) at runtime.
|
||||||
|
// the MMVQ batch size check is done in ggml_cuda_try_fuse
|
||||||
if (!is_pair(gate, up, routed_idx) || !is_pair(shared_gate, shared_up, shared_idx) ||
|
if (!is_pair(gate, up, routed_idx) || !is_pair(shared_gate, shared_up, shared_idx) ||
|
||||||
!ggml_cuda_should_fuse_mul_mat(up, gate, routed) ||
|
!ggml_cuda_should_fuse_mul_mat(up, gate, routed) ||
|
||||||
!ggml_cuda_should_fuse_mul_mat(shared_up, shared_gate, shared) ||
|
!ggml_cuda_should_fuse_mul_mat(shared_up, shared_gate, shared) ||
|
||||||
!up->src[0]->buffer ||
|
!up->src[0]->buffer ||
|
||||||
!ggml_cuda_should_fuse_mul_mat_vec_q(up)) {
|
!ggml_is_quantized(up->src[0]->type)) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
const ggml_tensor * input = up->src[1];
|
const ggml_tensor * input = up->src[1];
|
||||||
@@ -3510,7 +3513,8 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
|
|||||||
ggml_tensor * node = cgraph->nodes[i];
|
ggml_tensor * node = cgraph->nodes[i];
|
||||||
|
|
||||||
if (node->op == GGML_OP_MUL_MAT_ID && cuda_ctx->stream_context().concurrent_events.empty() &&
|
if (node->op == GGML_OP_MUL_MAT_ID && cuda_ctx->stream_context().concurrent_events.empty() &&
|
||||||
ggml_cuda_match_shared_expert(cgraph, i, i + 3)) {
|
ggml_cuda_match_shared_expert(cgraph, i, i + 3) &&
|
||||||
|
ggml_cuda_should_fuse_mul_mat_vec_q(cgraph->nodes[i + 2]->src[1])) {
|
||||||
const int outputs[] = { i + 2, i + 5 };
|
const int outputs[] = { i + 2, i + 5 };
|
||||||
if (ggml_cuda_check_fusion_memory_ranges(cgraph, i, 6, outputs, 2)) {
|
if (ggml_cuda_check_fusion_memory_ranges(cgraph, i, 6, outputs, 2)) {
|
||||||
ggml_tensor * routed = cgraph->nodes[i + 2];
|
ggml_tensor * routed = cgraph->nodes[i + 2];
|
||||||
|
|||||||
Reference in New Issue
Block a user