mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-11 07:20:33 +02:00
meta : handle host views (#30217)
* meta : handle views of tensors allocated on the host A view shares the memory of its view_src, so ggml-alloc never allocates a view in the buffer of the split it lands in - the scheduler copies the source into the split and the ops that use the view read that copy. The view node itself is a noop and does not need a split of its own, but the meta backend asserted when one was left inside a meta split: - ggml_backend_meta_get_split_state() dereferenced tensor->buffer->context - the graph rebuild mapped every node with ggml_backend_meta_buffer_simple_tensor() Accept such nodes when they are views of host tensors, which also generalizes the previous s_copy_main workaround. This fixes the assert hit by KV cache views when using --split-mode tensor with partial offload. Assisted-by: pi:llama.cpp/Qwen3.8-Flash-Next * archs : re-enable sm tensor for K2 Horizon * cont : add TODO and reference
This commit is contained in:
@@ -1176,7 +1176,22 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(
|
||||
return ret;
|
||||
}
|
||||
|
||||
static bool ggml_backend_meta_is_host_view(const struct ggml_tensor * tensor) {
|
||||
return ggml_is_view(tensor) && ggml_backend_buffer_is_host(tensor->view_src->buffer);
|
||||
}
|
||||
|
||||
static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(const struct ggml_tensor * tensor, bool assume_sync) {
|
||||
// [TAG_META_HOST_VIEWS]
|
||||
// TODO: technically, this check should not be needed if the backend scheduler correctly prevents assigning
|
||||
// such host-buffer views to the meta backend. figure out how to update the scheduler logic to achieve that
|
||||
// ref: https://github.com/ggml-org/llama.cpp/pull/30217
|
||||
if (!ggml_backend_buffer_is_meta(tensor->buffer)) {
|
||||
GGML_ASSERT(ggml_backend_meta_is_host_view(tensor));
|
||||
|
||||
// the view is not allocated in the meta buffer, it is not split across the sub-devices
|
||||
return { GGML_BACKEND_SPLIT_AXIS_NONE, {0}, {1}, 1 };
|
||||
}
|
||||
|
||||
ggml_backend_meta_buffer_context * buf_ctx = (ggml_backend_meta_buffer_context *) tensor->buffer->context;
|
||||
return ggml_backend_meta_get_split_state(buf_ctx->get_simple_tensor_container(tensor), tensor, assume_sync);
|
||||
}
|
||||
@@ -2026,9 +2041,11 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend,
|
||||
|
||||
for (int i = 0; i < cgraph->n_nodes; i++) {
|
||||
ggml_tensor * node = cgraph->nodes[i];
|
||||
if (node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && ggml_backend_buffer_is_host(node->view_src->buffer)) {
|
||||
// FIXME s_copy_main is on the CPU and its view seems to be incorrectly added to the graph nodes.
|
||||
// For regular usage this doesn't matter since it's a noop but trying to call ggml_backend_meta_buffer_simple_tensor results in a crash.
|
||||
if (!ggml_backend_buffer_is_meta(node->buffer)) {
|
||||
// [TAG_META_HOST_VIEWS]
|
||||
GGML_ASSERT(ggml_backend_meta_is_host_view(node));
|
||||
|
||||
// keep the node as is, mapping it to a simple tensor is not possible
|
||||
bcj.nodes[i] = node;
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -1234,7 +1234,6 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {
|
||||
case LLM_ARCH_KIMI_K3:
|
||||
case LLM_ARCH_GLM5_NEXT:
|
||||
case LLM_ARCH_QWEN3TTS:
|
||||
case LLM_ARCH_K2_HORIZON:
|
||||
return false;
|
||||
default:
|
||||
return true;
|
||||
|
||||
Reference in New Issue
Block a user