meta : handle host views (#30217)

* meta : handle views of tensors allocated on the host

A view shares the memory of its view_src, so ggml-alloc never allocates a view in
the buffer of the split it lands in - the scheduler copies the source into the
split and the ops that use the view read that copy. The view node itself is a noop
and does not need a split of its own, but the meta backend asserted when one was
left inside a meta split:

  - ggml_backend_meta_get_split_state() dereferenced tensor->buffer->context
  - the graph rebuild mapped every node with ggml_backend_meta_buffer_simple_tensor()

Accept such nodes when they are views of host tensors, which also generalizes the
previous s_copy_main workaround. This fixes the assert hit by KV cache views when
using --split-mode tensor with partial offload.

Assisted-by: pi:llama.cpp/Qwen3.8-Flash-Next

* archs : re-enable sm tensor for K2 Horizon

* cont : add TODO and reference
This commit is contained in:
Georgi Gerganov
2026-10-09 16:01:04 +03:00
committed by GitHub
parent ba6439a6b5
commit e60eff95fd
2 changed files with 20 additions and 4 deletions
+20 -3
View File
@@ -1176,7 +1176,22 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(
return ret;
}
static bool ggml_backend_meta_is_host_view(const struct ggml_tensor * tensor) {
return ggml_is_view(tensor) && ggml_backend_buffer_is_host(tensor->view_src->buffer);
}
static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(const struct ggml_tensor * tensor, bool assume_sync) {
// [TAG_META_HOST_VIEWS]
// TODO: technically, this check should not be needed if the backend scheduler correctly prevents assigning
// such host-buffer views to the meta backend. figure out how to update the scheduler logic to achieve that
// ref: https://github.com/ggml-org/llama.cpp/pull/30217
if (!ggml_backend_buffer_is_meta(tensor->buffer)) {
GGML_ASSERT(ggml_backend_meta_is_host_view(tensor));
// the view is not allocated in the meta buffer, it is not split across the sub-devices
return { GGML_BACKEND_SPLIT_AXIS_NONE, {0}, {1}, 1 };
}
ggml_backend_meta_buffer_context * buf_ctx = (ggml_backend_meta_buffer_context *) tensor->buffer->context;
return ggml_backend_meta_get_split_state(buf_ctx->get_simple_tensor_container(tensor), tensor, assume_sync);
}
@@ -2026,9 +2041,11 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend,
for (int i = 0; i < cgraph->n_nodes; i++) {
ggml_tensor * node = cgraph->nodes[i];
if (node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && ggml_backend_buffer_is_host(node->view_src->buffer)) {
// FIXME s_copy_main is on the CPU and its view seems to be incorrectly added to the graph nodes.
// For regular usage this doesn't matter since it's a noop but trying to call ggml_backend_meta_buffer_simple_tensor results in a crash.
if (!ggml_backend_buffer_is_meta(node->buffer)) {
// [TAG_META_HOST_VIEWS]
GGML_ASSERT(ggml_backend_meta_is_host_view(node));
// keep the node as is, mapping it to a simple tensor is not possible
bcj.nodes[i] = node;
continue;
}
-1
View File
@@ -1234,7 +1234,6 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {
case LLM_ARCH_KIMI_K3:
case LLM_ARCH_GLM5_NEXT:
case LLM_ARCH_QWEN3TTS:
case LLM_ARCH_K2_HORIZON:
return false;
default:
return true;