Commit e60eff95f for llama.cpp

commit e60eff95fda7b6e46bf402a8e9f1e86d88ab29c8
Author: Georgi Gerganov <ggerganov@gmail.com>
Date:   Fri Oct 9 16:01:04 2026 +0300

    meta : handle host views (#30217)

    * meta : handle views of tensors allocated on the host

    A view shares the memory of its view_src, so ggml-alloc never allocates a view in
    the buffer of the split it lands in - the scheduler copies the source into the
    split and the ops that use the view read that copy. The view node itself is a noop
    and does not need a split of its own, but the meta backend asserted when one was
    left inside a meta split:

      - ggml_backend_meta_get_split_state() dereferenced tensor->buffer->context
      - the graph rebuild mapped every node with ggml_backend_meta_buffer_simple_tensor()

    Accept such nodes when they are views of host tensors, which also generalizes the
    previous s_copy_main workaround. This fixes the assert hit by KV cache views when
    using --split-mode tensor with partial offload.

    Assisted-by: pi:llama.cpp/Qwen3.8-Flash-Next

    * archs : re-enable sm tensor for K2 Horizon

    * cont : add TODO and reference

diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp
index 8ed5f4ebb..86748b689 100644
--- a/ggml/src/ggml-backend-meta.cpp
+++ b/ggml/src/ggml-backend-meta.cpp
@@ -1176,7 +1176,22 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(
     return ret;
 }

+static bool ggml_backend_meta_is_host_view(const struct ggml_tensor * tensor) {
+    return ggml_is_view(tensor) && ggml_backend_buffer_is_host(tensor->view_src->buffer);
+}
+
 static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(const struct ggml_tensor * tensor, bool assume_sync) {
+    // [TAG_META_HOST_VIEWS]
+    // TODO: technically, this check should not be needed if the backend scheduler correctly prevents assigning
+    //       such host-buffer views to the meta backend. figure out how to update the scheduler logic to achieve that
+    //       ref: https://github.com/ggml-org/llama.cpp/pull/30217
+    if (!ggml_backend_buffer_is_meta(tensor->buffer)) {
+        GGML_ASSERT(ggml_backend_meta_is_host_view(tensor));
+
+        // the view is not allocated in the meta buffer, it is not split across the sub-devices
+        return { GGML_BACKEND_SPLIT_AXIS_NONE, {0}, {1}, 1 };
+    }
+
     ggml_backend_meta_buffer_context * buf_ctx = (ggml_backend_meta_buffer_context *) tensor->buffer->context;
     return ggml_backend_meta_get_split_state(buf_ctx->get_simple_tensor_container(tensor), tensor, assume_sync);
 }
@@ -2026,9 +2041,11 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend,

             for (int i = 0; i < cgraph->n_nodes; i++) {
                 ggml_tensor * node = cgraph->nodes[i];
-                if (node->view_src != nullptr && node->view_src->op == GGML_OP_NONE && ggml_backend_buffer_is_host(node->view_src->buffer)) {
-                    // FIXME s_copy_main is on the CPU and its view seems to be incorrectly added to the graph nodes.
-                    // For regular usage this doesn't matter since it's a noop but trying to call ggml_backend_meta_buffer_simple_tensor results in a crash.
+                if (!ggml_backend_buffer_is_meta(node->buffer)) {
+                    // [TAG_META_HOST_VIEWS]
+                    GGML_ASSERT(ggml_backend_meta_is_host_view(node));
+
+                    // keep the node as is, mapping it to a simple tensor is not possible
                     bcj.nodes[i] = node;
                     continue;
                 }
diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
index 821479064..5774867d9 100644
--- a/src/llama-arch.cpp
+++ b/src/llama-arch.cpp
@@ -1234,7 +1234,6 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {
         case LLM_ARCH_KIMI_K3:
         case LLM_ARCH_GLM5_NEXT:
         case LLM_ARCH_QWEN3TTS:
-        case LLM_ARCH_K2_HORIZON:
             return false;
         default:
             return true;