Commit d6cf9acb2 for llama.cpp
commit d6cf9acb25e6c657d6c6a4645ab7b69fce119995
Author: Aman Gupta <amangupta052@gmail.com>
Date: Wed Oct 7 23:37:45 2026 +0530
llama : add a GPU cache for MoE experts kept in host memory (#29887)
* llama : add a GPU cache for MoE experts kept in host memory
Assisted-by: Claude
* use llama_moe_cache_ptr
diff --git a/common/arg.cpp b/common/arg.cpp
index 14d82f695..182f76f45 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -2776,6 +2776,16 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
llm_add_n_cpu_ffn_overrides(value, LLM_FFN_EXPS_REGEX, params.tensor_buft_overrides);
}
).set_env("LLAMA_ARG_N_CPU_MOE"));
+ add_opt(common_arg(
+ {"--moe-cache-mib"}, "N",
+ "GPU cache size in MiB for the MoE experts kept in the CPU (default: 0, disabled)",
+ [](common_params & params, int value) {
+ if (value < 0) {
+ throw std::invalid_argument("invalid value");
+ }
+ params.moe_cache_size = (size_t) value*1024*1024;
+ }
+ ).set_env("LLAMA_ARG_MOE_CACHE_MIB"));
add_opt(common_arg(
{"-ncffn", "--n-cpu-ffn"}, "N",
"keep the dense FFN weights of the first N layers in the CPU\n"
diff --git a/common/common.cpp b/common/common.cpp
index e36f8ab50..768b2e9a5 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1722,6 +1722,8 @@ struct llama_context_params common_context_params_to_llama(const common_params &
cparams.type_k = params.cache_type_k;
cparams.type_v = params.cache_type_v;
+ cparams.moe_cache_size = params.moe_cache_size;
+
return cparams;
}
diff --git a/common/common.h b/common/common.h
index 3e3eff379..0f912d901 100644
--- a/common/common.h
+++ b/common/common.h
@@ -593,6 +593,8 @@ struct common_params {
ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K
ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V
+ size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU
+
common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;
// multimodal models (see tools/mtmd)
diff --git a/common/speculative.cpp b/common/speculative.cpp
index a7f095a12..d9ddf44d2 100644
--- a/common/speculative.cpp
+++ b/common/speculative.cpp
@@ -2561,6 +2561,9 @@ common_params common_base_params_to_speculative(const common_params & params) {
result.n_outputs_max = params.n_parallel;
result.n_outputs_max_per_seq = 1;
+ // the MoE cache is only used by the target context
+ result.moe_cache_size = 0;
+
// dflash/dspark decode the whole noise block in a single pass and sample every block position on the backend
// TODO: refactor such properties to be announced by the speculative types
// something like `struct common_speculative_type_props common_speculative_type_get_props(...);`
diff --git a/include/llama.h b/include/llama.h
index 260247e82..77f527d5b 100644
--- a/include/llama.h
+++ b/include/llama.h
@@ -396,6 +396,8 @@ extern "C" {
enum ggml_type type_k; // data type for K cache [EXPERIMENTAL]
enum ggml_type type_v; // data type for V cache [EXPERIMENTAL]
+ size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, 0 = disabled [EXPERIMENTAL]
+
// Abort callback
// if it returns true, execution of llama_decode() will be aborted
// currently works only with CPU execution
diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt
index afdaddc79..97250c2e8 100644
--- a/src/CMakeLists.txt
+++ b/src/CMakeLists.txt
@@ -28,6 +28,7 @@ set(LLAMA_CORE_SOURCES
llama-kv-cache-msa.cpp
llama-kv-cache-dsv4.cpp
llama-memory.cpp
+ llama-moe-cache.cpp
llama-memory-hybrid.cpp
llama-memory-hybrid-iswa.cpp
llama-memory-hybrid-idx.cpp
diff --git a/src/llama-context.cpp b/src/llama-context.cpp
index ff2ea461c..5f861f96c 100644
--- a/src/llama-context.cpp
+++ b/src/llama-context.cpp
@@ -9,6 +9,7 @@
#include "llama-memory.h"
#include "llama-mmap.h"
#include "llama-model.h"
+#include "llama-moe-cache.h"
#include "llama-ext.h"
#include "llama-sampler.h"
#include "llama.h"
@@ -272,8 +273,9 @@ llama_context::llama_context(
}
}
- cparams.op_offload = params.op_offload;
- cparams.kv_unified = params.kv_unified;
+ cparams.op_offload = params.op_offload;
+ cparams.kv_unified = params.kv_unified;
+ cparams.moe_cache_size = params.moe_cache_size;
// initialized later
cparams.pipeline_parallel = false;
@@ -462,6 +464,22 @@ llama_context::llama_context(
LLAMA_LOG_INFO("%s: pipeline parallelism enabled\n", __func__);
}
+ if (cparams.moe_cache_size > 0) {
+ if (cparams.pipeline_parallel || model.n_devices() > 1) {
+ throw std::runtime_error("MoE cache does not support multiple devices");
+ }
+ for (size_t i = 0; i < backend_ptrs.size(); ++i) {
+ const auto type = ggml_backend_dev_type(ggml_backend_get_device(backend_ptrs[i]));
+ if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
+ moe_cache = std::make_unique<llama_moe_cache>(model, backend_ptrs[i], backend_buft[i], cparams.moe_cache_size);
+ break;
+ }
+ }
+ if (!moe_cache) {
+ throw std::runtime_error("MoE cache requires a GPU backend");
+ }
+ }
+
sched_reserve();
if (!cparams.flash_attn) {
@@ -2605,6 +2623,7 @@ llm_graph_params llama_context::graph_params(
/*.loras =*/ loras.get(),
/*.mctx =*/ mctx,
/*.cross =*/ &cross,
+ /*.moe_cache =*/ moe_cache.get(),
/*.prec_policy =*/ &model.prec_policy,
/*.samplers =*/ sampling.samplers,
/*.n_outputs =*/ n_outputs,
@@ -2645,7 +2664,14 @@ ggml_status llama_context::graph_compute(
}
bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data) {
- auto & st = static_cast<llama_context *>(user_data)->copy_experts;
+ auto * lctx = static_cast<llama_context *>(user_data);
+
+ // the slot maps of the MoE cache
+ if (lctx->moe_cache && lctx->moe_cache->copy(backend, src, dst, graph)) {
+ return true;
+ }
+
+ auto & st = lctx->copy_experts;
// the ids must be computed before the split starts, so only the first node of the split is considered
if (ggml_graph_n_nodes(graph) == 0) {
@@ -2692,10 +2718,28 @@ bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor
last++;
}
+ // the experts in the MoE cache are copied from device memory, the others are uploaded
+ int64_t next = first;
+ for (int64_t e = first; e <= last && lctx->moe_cache; ) {
+ const int64_t n = lctx->moe_cache->copy_experts(backend, src, dst, e, last);
+ if (n == 0) {
+ e++;
+ continue;
+ }
+ if (next < e) {
+ ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + next*expert_size, next*expert_size, (e - next)*expert_size);
+ }
+ e += n;
+ next = e;
+ }
+
// copy a bit extra to ensure there are no NaNs in the padding of the last expert, this is necessary for MMQ in the CUDA backend
- const size_t offset = first*expert_size;
+ const size_t offset = next*expert_size;
const size_t padding = last < n_expert - 1 ? std::min<size_t>(expert_size, 512) : 0;
- ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, (last - first + 1)*expert_size + padding);
+ const size_t size = (last + 1 - next)*expert_size + padding;
+ if (size > 0) {
+ ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, size);
+ }
first = last + 1;
}
@@ -3562,6 +3606,11 @@ llama_memory_breakdown llama_context::memory_breakdown() const {
ret[buft].context += size;
}
}
+ if (moe_cache) {
+ for (const auto & [buft, size] : moe_cache->memory_breakdown()) {
+ ret[buft].context += size;
+ }
+ }
if (model.hparams.no_alloc) {
for (size_t i = 0; i < backends.size(); ++i) {
ggml_backend_t backend = backends[i].get();
@@ -3851,6 +3900,7 @@ llama_context_params llama_context_default_params() {
/*.cb_eval_user_data =*/ nullptr,
/*.type_k =*/ GGML_TYPE_F16,
/*.type_v =*/ GGML_TYPE_F16,
+ /*.moe_cache_size =*/ 0,
/*.abort_callback =*/ nullptr,
/*.abort_callback_data =*/ nullptr,
/*.embeddings =*/ false,
diff --git a/src/llama-context.h b/src/llama-context.h
index 28d386ad7..69bc01c19 100644
--- a/src/llama-context.h
+++ b/src/llama-context.h
@@ -7,6 +7,7 @@
#include "llama-adapter.h"
#include "llama-impl.h"
#include "llama-memory.h"
+#include "llama-moe-cache.h"
#include "ggml-cpp.h"
#include "ggml-opt.h"
@@ -17,6 +18,7 @@
struct llama_model;
class llama_batch_allocr;
+class llama_moe_cache;
class llama_io_read_i;
class llama_io_write_i;
@@ -271,7 +273,7 @@ private:
llm_graph_cb graph_get_cb() const;
- // ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID
+ // ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID and updates the MoE cache
static bool sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data);
// disable auto fused ops (Flash Attention, Gated Delta Net) whose op lands on a device
@@ -299,6 +301,7 @@ private:
llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably
llama_memory_ptr memory;
+ llama_moe_cache_ptr moe_cache;
// decode output (2-dimensional array: [n_outputs][n_vocab])
buffer_view<float> logits = {nullptr, 0};
diff --git a/src/llama-cparams.h b/src/llama-cparams.h
index 004fec5a6..f4eb181a7 100644
--- a/src/llama-cparams.h
+++ b/src/llama-cparams.h
@@ -55,6 +55,8 @@ struct llama_cparams {
bool pipeline_parallel;
bool training; // set by llama_opt_init()
+ size_t moe_cache_size;
+
std::vector<bool> embeddings_layer_inp; // [n_layer()] extract input embeddings for layer
enum llama_context_type ctx_type;
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index 1112ad885..e0b7a47a9 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -2,6 +2,7 @@
#include "llama-impl.h"
#include "llama-model.h"
+#include "llama-moe-cache.h"
#include "llama-batch.h"
#include "llama-cparams.h"
#include "llama-sampler.h"
@@ -1523,6 +1524,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) :
loras (params.loras),
mctx (params.mctx),
cross (params.cross),
+ moe_cache (params.moe_cache),
prec_policy (params.prec_policy),
samplers (params.samplers),
cb_func (params.cb),
@@ -1589,8 +1591,12 @@ ggml_tensor * llm_graph_context::build_lora_mm_id(
ggml_tensor * w, // ggml_tensor * as
ggml_tensor * cur, // ggml_tensor * b
ggml_tensor * ids,
- ggml_tensor * w_s) const {
- ggml_tensor * res = ggml_mul_mat_id(ctx0, w, cur, ids);
+ ggml_tensor * w_s,
+ ggml_tensor * slots) const {
+ // the experts in the MoE cache are selected by their slots
+ ggml_tensor * res = slots == nullptr ?
+ ggml_mul_mat_id(ctx0, w, cur, ids) :
+ ggml_mul_mat_id(ctx0, moe_cache->get_experts(w), cur, slots);
if (prec_policy) {
prec_policy->apply(res);
@@ -2205,6 +2211,9 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
//call early so that topk-moe can be used
ggml_build_forward_expand(gf, weights);
+ // the experts of host-resident layers may be read from the MoE cache
+ ggml_tensor * slots = build_moe_cache_slots(selected_experts, up_exps, gate_exps, down_exps, gate_up_exps, il);
+
cur = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
if (weight_before_ffn) {
@@ -2219,7 +2228,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
if (gate_up_exps) {
// merged gate_up path: one mul_mat_id, then split into gate and up views
- ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s); // [n_ff*2, n_expert_used, n_tokens]
+ ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff*2, n_expert_used, n_tokens]
cb(gate_up, "ffn_moe_gate_up", il);
if (up_exps_s) {
@@ -2238,7 +2247,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
cb(up, "ffn_moe_up", il);
} else {
// separate gate and up path
- up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s); // [n_ff, n_expert_used, n_tokens]
+ up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff, n_expert_used, n_tokens]
cb(up, "ffn_moe_up", il);
if (up_exps_s) {
@@ -2251,7 +2260,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
}
if (gate_exps) {
- cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s); // [n_ff, n_expert_used, n_tokens]
+ cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s, slots); // [n_ff, n_expert_used, n_tokens]
cb(cur, "ffn_moe_gate", il);
} else {
cur = up;
@@ -2352,7 +2361,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
GGML_ABORT("fatal error");
}
- experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s); // [n_embd, n_expert_used, n_tokens]
+ experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s, slots); // [n_embd, n_expert_used, n_tokens]
if (arch == LLM_ARCH_MISTRAL4) {
// src1 can exceed F16 range
ggml_prec_set_src(experts, GGML_PREC_F32, 1);
@@ -2409,6 +2418,45 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
return moe_out;
}
+ggml_tensor * llm_graph_context::build_moe_cache_slots(
+ ggml_tensor * selected_experts,
+ ggml_tensor * up_exps,
+ ggml_tensor * gate_exps,
+ ggml_tensor * down_exps,
+ ggml_tensor * gate_up_exps,
+ int il) const {
+ if (moe_cache == nullptr) {
+ return nullptr;
+ }
+
+ ggml_tensor * slot_map = moe_cache->get_slot_map(il, selected_experts->ne[1], selected_experts->ne[0]);
+ if (slot_map == nullptr) {
+ return nullptr;
+ }
+ for (ggml_tensor * w : { up_exps, gate_exps, down_exps, gate_up_exps }) {
+ if (w != nullptr && moe_cache->get_experts(w) == nullptr) {
+ return nullptr;
+ }
+ }
+
+ ggml_tensor * ids = selected_experts;
+ if (!ggml_is_contiguous(ids)) {
+ ids = ggml_cont(ctx0, ids);
+ }
+ ids = ggml_reshape_1d(ctx0, ids, ggml_nelements(ids));
+
+ // the slot map is a host weight, so the scheduler starts a new split here and copies it with the copy callback
+ // the callback reads the selected experts, uploads the missing ones and updates the slot map
+ ggml_tensor * slots = ggml_get_rows(ctx0, slot_map, ids); // [1, n_expert_used*n_tokens]
+ if (!ggml_backend_supports_op(moe_cache->backend(), slots)) {
+ return nullptr;
+ }
+ ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend());
+ cb(slots, "ffn_moe_slots", il);
+
+ return ggml_reshape_2d(ctx0, slots, selected_experts->ne[0], selected_experts->ne[1]); // [n_expert_used, n_tokens]
+}
+
// input embeddings with optional lora
ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float tok_scale) const {
const int64_t n_embd_inp = hparams.n_embd_inp();
diff --git a/src/llama-graph.h b/src/llama-graph.h
index 838544576..79e8409ac 100644
--- a/src/llama-graph.h
+++ b/src/llama-graph.h
@@ -21,6 +21,8 @@ struct llama_cparams;
struct llama_layer;
struct llama_prec_policy;
+class llama_moe_cache;
+
struct llama_memory_context_i;
class llama_kv_cache_context;
@@ -793,6 +795,7 @@ struct llm_graph_params {
const llama_adapter_loras * loras;
const llama_memory_context_i * mctx;
const llama_cross * cross;
+ const llama_moe_cache * moe_cache;
const llama_prec_policy * prec_policy = nullptr;
@@ -1036,6 +1039,7 @@ struct llm_graph_context {
const llama_adapter_loras * loras;
const llama_memory_context_i * mctx;
const llama_cross * cross;
+ const llama_moe_cache * moe_cache;
const llama_prec_policy * prec_policy;
@@ -1078,11 +1082,13 @@ struct llm_graph_context {
ggml_tensor * w_s = nullptr) const;
// do mat_mul_id, while optionally apply lora and per-expert scale
+ // if slots is set, the experts are read from the MoE cache at these slots (see build_moe_cache_slots)
ggml_tensor * build_lora_mm_id(
ggml_tensor * w, // ggml_tensor * as
ggml_tensor * cur, // ggml_tensor * b
ggml_tensor * ids,
- ggml_tensor * w_s = nullptr) const;
+ ggml_tensor * w_s = nullptr,
+ ggml_tensor * slots = nullptr) const;
ggml_tensor * build_norm(
ggml_tensor * cur,
@@ -1179,6 +1185,15 @@ struct llm_graph_context {
ggml_tensor * down_exps_s = nullptr,
ggml_tensor * selected_experts_in = nullptr) const;
+ // the slots of the selected experts in the MoE cache, nullptr if the experts of the layer are not read from the cache
+ ggml_tensor * build_moe_cache_slots(
+ ggml_tensor * selected_experts,
+ ggml_tensor * up_exps,
+ ggml_tensor * gate_exps,
+ ggml_tensor * down_exps,
+ ggml_tensor * gate_up_exps,
+ int il) const;
+
//
// inputs
//
diff --git a/src/llama-moe-cache.cpp b/src/llama-moe-cache.cpp
new file mode 100644
index 000000000..7292a1e57
--- /dev/null
+++ b/src/llama-moe-cache.cpp
@@ -0,0 +1,550 @@
+#include "llama-moe-cache.h"
+
+#include "llama-impl.h"
+#include "llama-model.h"
+
+#include "ggml-cpp.h"
+
+#include <algorithm>
+#include <stdexcept>
+#include <unordered_map>
+#include <vector>
+
+namespace {
+
+// LRU of the experts of a group of layers, the slot of each expert is kept in the slot map of its layer
+struct moe_cache_lru {
+ int32_t n_expert = 0;
+ int32_t n_slots = 0;
+
+ std::vector<int32_t *> slot_map; // [n_layer] data of the slot maps, -1 if the expert is not cached
+ std::vector<int32_t> key_of; // [n_slots] il*n_expert + expert, -1 if empty
+
+ // doubly linked list of the slots, head is the least recently used
+ std::vector<int32_t> prev;
+ std::vector<int32_t> next;
+ int32_t head = -1;
+ int32_t tail = -1;
+
+ std::vector<uint32_t> seen; // [n_expert]
+ uint32_t seen_gen = 0;
+ std::vector<int32_t> uniq;
+
+ void init(int32_t n_layer, int32_t n_expert, int32_t n_slots) {
+ this->n_expert = n_expert;
+ this->n_slots = n_slots;
+ slot_map.assign(n_layer, nullptr);
+ key_of.assign(n_slots, -1);
+ prev.resize(n_slots);
+ next.resize(n_slots);
+ for (int32_t s = 0; s < n_slots; ++s) {
+ prev[s] = s - 1;
+ next[s] = s + 1 < n_slots ? s + 1 : -1;
+ }
+ head = 0;
+ tail = n_slots - 1;
+ seen.assign(n_expert, 0);
+ }
+
+ // move slot s to the tail (most recently used)
+ void touch(int32_t s) {
+ if (s == tail) {
+ return;
+ }
+ if (prev[s] >= 0) {
+ next[prev[s]] = next[s];
+ } else {
+ head = next[s];
+ }
+ prev[next[s]] = prev[s];
+
+ prev[s] = tail;
+ next[s] = -1;
+ next[tail] = s;
+ tail = s;
+ }
+
+ struct fill {
+ int32_t expert;
+ int32_t slot;
+ };
+
+ // give a slot to each expert selected by ids in layer il, the misses evict the least recently used experts
+ // returns false if the ids select more distinct experts than there are slots
+ bool plan(int32_t il, const int32_t * ids, size_t n_ids, std::vector<fill> & fills, size_t & n_hit) {
+ fills.clear();
+ n_hit = 0;
+
+ if (++seen_gen == 0) {
+ std::fill(seen.begin(), seen.end(), 0);
+ seen_gen = 1;
+ }
+ uniq.clear();
+ for (size_t i = 0; i < n_ids; ++i) {
+ GGML_ASSERT(ids[i] >= 0 && ids[i] < n_expert);
+ if (seen[ids[i]] != seen_gen) {
+ seen[ids[i]] = seen_gen;
+ uniq.push_back(ids[i]);
+ }
+ }
+ if (uniq.size() > (size_t) n_slots) {
+ return false;
+ }
+
+ int32_t * slots = slot_map[il];
+
+ // hits go to the tail first, so the head can be evicted below
+ for (int32_t e : uniq) {
+ if (slots[e] >= 0) {
+ touch(slots[e]);
+ n_hit++;
+ }
+ }
+ // sorted misses usually get consecutive slots, so the uploads can be merged
+ std::sort(uniq.begin(), uniq.end());
+ for (int32_t e : uniq) {
+ if (slots[e] >= 0) {
+ continue;
+ }
+ const int32_t s = head;
+ if (key_of[s] >= 0) {
+ slot_map[key_of[s] / n_expert][key_of[s] % n_expert] = -1;
+ }
+ key_of[s] = il*n_expert + e;
+ slots[e] = s;
+ touch(s);
+ fills.push_back({ e, s });
+ }
+ return true;
+ }
+};
+
+// gate, up, down or gate_up, down
+static std::vector<ggml_tensor *> llama_moe_cache_layer_experts(const llama_layer & layer) {
+ std::vector<ggml_tensor *> res;
+ for (ggml_tensor * t : { layer.ffn_gate_up_exps, layer.ffn_gate_exps, layer.ffn_up_exps, layer.ffn_down_exps }) {
+ if (t != nullptr) {
+ res.push_back(t);
+ }
+ }
+ return res;
+}
+
+static bool llama_moe_cache_same_layout(const std::vector<ggml_tensor *> & a, const std::vector<ggml_tensor *> & b) {
+ if (a.size() != b.size()) {
+ return false;
+ }
+ for (size_t i = 0; i < a.size(); ++i) {
+ if (a[i]->type != b[i]->type || !ggml_are_same_shape(a[i], b[i]) || a[i]->nb[2] != b[i]->nb[2]) {
+ return false;
+ }
+ }
+ return true;
+}
+
+static bool llama_moe_cache_is_host_weight(const ggml_tensor * t) {
+ return t->buffer != nullptr &&
+ ggml_backend_buffer_get_usage(t->buffer) == GGML_BACKEND_BUFFER_USAGE_WEIGHTS &&
+ ggml_backend_buffer_is_host(t->buffer);
+}
+
+}
+
+struct llama_moe_cache::impl {
+ // layers with the same expert tensor layout share the banks and the LRU of a group
+ struct group {
+ std::vector<ggml_tensor *> ref; // expert tensors of the first layer
+ std::vector<int32_t> layers;
+ std::vector<ggml_tensor *> banks; // device storage of all slots, one per expert tensor
+ size_t host_bytes = 0;
+ int32_t n_slots = 0;
+ moe_cache_lru lru;
+ };
+
+ struct layer {
+ int32_t ig = -1; // -1 if the layer is not cached
+ ggml_tensor * slot_map = nullptr; // I32 [1, n_expert] in host memory
+ std::vector<ggml_tensor *> experts; // host expert tensors, in the order of the banks
+ };
+
+ struct binding {
+ int32_t il;
+ int32_t ip; // index of the bank
+ ggml_tensor * cached; // view of the bank used in place of the host experts
+ };
+
+ struct stats {
+ size_t hits = 0;
+ size_t misses = 0;
+ size_t bytes = 0;
+ };
+
+ static constexpr int64_t max_batch = 32;
+
+ ggml_backend_t backend;
+ int32_t n_expert_used;
+
+ stats stats_small; // up to 8 tokens per ubatch
+ stats stats_large;
+ stats stats_copy; // experts copied from the cache for large batches
+
+ std::vector<group> groups;
+ std::vector<layer> layers;
+ std::unordered_map<const ggml_tensor *, binding> bindings; // host experts -> cached experts
+ std::unordered_map<const ggml_tensor *, int32_t> layer_of; // slot map -> layer
+
+ std::vector<int32_t> ids;
+ std::vector<moe_cache_lru::fill> fills;
+
+ // banks and their views on the device
+ ggml_context_ptr ctx;
+ ggml_backend_buffer_ptr buf;
+ size_t buf_size = 0;
+
+ // slot maps in host memory
+ ggml_context_ptr ctx_host;
+ ggml_backend_buffer_ptr buf_host;
+ size_t buf_host_size = 0;
+
+ // views used by copy_experts
+ ggml_context_ptr ctx_views;
+
+ impl(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
+ backend(backend), n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) {
+ ggml_backend_dev_t dev = ggml_backend_get_device(backend);
+ const auto dev_type = ggml_backend_dev_type(dev);
+ if (dev_type != GGML_BACKEND_DEVICE_TYPE_GPU && dev_type != GGML_BACKEND_DEVICE_TYPE_IGPU) {
+ throw std::runtime_error("MoE cache requires a GPU backend");
+ }
+ if (model.split_mode() == LLAMA_SPLIT_MODE_TENSOR) {
+ throw std::runtime_error("MoE cache does not support tensor parallelism");
+ }
+ if (model.hparams.n_expert == 0 || n_expert_used == 0) {
+ throw std::runtime_error("MoE cache requires a MoE model");
+ }
+
+ // only cache layers that keep all of their experts in host memory
+ size_t host_bytes = 0;
+ for (size_t il = 0; il < model.layers.size(); ++il) {
+ auto experts = llama_moe_cache_layer_experts(model.layers[il]);
+ if (experts.empty() || model.dev_layer(il) != dev ||
+ !std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) {
+ continue;
+ }
+ auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return llama_moe_cache_same_layout(g.ref, experts); });
+ if (it == groups.end()) {
+ groups.emplace_back();
+ it = groups.end() - 1;
+ it->ref = experts;
+ }
+ it->layers.push_back(il);
+ for (const ggml_tensor * t : experts) {
+ it->host_bytes += ggml_nbytes(t);
+ host_bytes += ggml_nbytes(t);
+ }
+ }
+ if (groups.empty()) {
+ LLAMA_LOG_WARN("%s: no layer has all of its experts in host memory, MoE cache is disabled\n", __func__);
+ return;
+ }
+
+ // one extra slot at the end, CUDA MMQ can read past the last expert
+ const size_t alignment = ggml_backend_buft_get_alignment(buft);
+ auto alloc_size = [&](const group & g, int32_t n_slots) {
+ size_t res = 0;
+ for (const ggml_tensor * t : g.ref) {
+ res += GGML_PAD(t->nb[2]*(n_slots + 1), alignment);
+ }
+ return res;
+ };
+
+ // split the budget by the size of the experts, so each group caches the same fraction of its experts
+ size_t n_tensors = 0;
+ size_t n_tensors_host = 0;
+ for (group & g : groups) {
+ const int32_t n_expert = g.ref[0]->ne[2];
+ const size_t budget = (size_t) ((double) size*g.host_bytes/host_bytes);
+ const int32_t max_slots = g.layers.size()*n_expert;
+ while (g.n_slots < max_slots && alloc_size(g, g.n_slots + 1) <= budget) {
+ g.n_slots++;
+ }
+ if (g.n_slots < n_expert_used) {
+ LLAMA_LOG_WARN("%s: MoE cache budget is too small for %zu layers, they are not cached\n", __func__, g.layers.size());
+ g.n_slots = 0;
+ continue;
+ }
+ g.lru.init(model.layers.size(), n_expert, g.n_slots);
+ n_tensors += g.ref.size()*(1 + g.layers.size());
+ n_tensors_host += g.layers.size();
+ }
+ if (n_tensors == 0) {
+ throw std::runtime_error("MoE cache is too small to hold the experts of one token");
+ }
+
+ auto init_ctx = [](size_t n_tensors) {
+ ggml_init_params params = {
+ /*.mem_size =*/ n_tensors*ggml_tensor_overhead(),
+ /*.mem_buffer =*/ nullptr,
+ /*.no_alloc =*/ true,
+ };
+ ggml_context_ptr res(ggml_init(params));
+ if (!res) {
+ throw std::runtime_error("failed to create the MoE cache context");
+ }
+ return res;
+ };
+ ctx = init_ctx(n_tensors);
+ ctx_host = init_ctx(n_tensors_host);
+ ctx_views = init_ctx(2);
+
+ ggml_backend_buffer_type_t buft_host = ggml_backend_cpu_buffer_type();
+ const size_t alignment_host = ggml_backend_buft_get_alignment(buft_host);
+
+ for (size_t ig = 0; ig < groups.size(); ++ig) {
+ group & g = groups[ig];
+ if (g.n_slots == 0) {
+ continue;
+ }
+ for (const ggml_tensor * t : g.ref) {
+ ggml_tensor * bank = ggml_new_tensor_3d(ctx.get(), t->type, t->ne[0], t->ne[1], g.n_slots + 1);
+ GGML_ASSERT(bank->nb[2] == t->nb[2]);
+ ggml_format_name(bank, "moe_cache.%zu.%s", ig, t->name);
+ g.banks.push_back(bank);
+ }
+ for (int32_t il : g.layers) {
+ layer & l = layers[il];
+ l.ig = (int32_t) ig;
+ l.experts = llama_moe_cache_layer_experts(model.layers[il]);
+ for (size_t ip = 0; ip < l.experts.size(); ++ip) {
+ ggml_tensor * bank = g.banks[ip];
+ ggml_tensor * cached = ggml_view_3d(ctx.get(), bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0);
+ ggml_format_name(cached, "moe_cache.%s", l.experts[ip]->name);
+ bindings[l.experts[ip]] = { il, (int32_t) ip, cached };
+ }
+ l.slot_map = ggml_new_tensor_2d(ctx_host.get(), GGML_TYPE_I32, 1, g.ref[0]->ne[2]);
+ ggml_format_name(l.slot_map, "moe_cache.slot_map-%d", il);
+ layer_of[l.slot_map] = il;
+ buf_host_size += GGML_PAD(ggml_nbytes(l.slot_map), alignment_host);
+ }
+ buf_size += alloc_size(g, g.n_slots);
+ }
+
+ if (model.hparams.no_alloc) {
+ // only used to measure the memory use, see llama_context::memory_breakdown
+ buf.reset(ggml_backend_buft_alloc_buffer(buft, 0));
+ buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0));
+ for (ggml_tensor * t = ggml_get_first_tensor(ctx.get()); t != nullptr; t = ggml_get_next_tensor(ctx.get(), t)) {
+ t->buffer = buf.get();
+ }
+ for (ggml_tensor * t = ggml_get_first_tensor(ctx_host.get()); t != nullptr; t = ggml_get_next_tensor(ctx_host.get(), t)) {
+ t->buffer = buf_host.get();
+ }
+ } else {
+ buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx.get(), buft));
+ buf_host.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx_host.get(), buft_host));
+ if (!buf || !buf_host) {
+ throw std::runtime_error("failed to allocate the MoE cache buffers");
+ }
+ ggml_backend_buffer_clear(buf.get(), 0);
+ ggml_backend_buffer_clear(buf_host.get(), 0xff); // all slots are -1
+ buf_size = ggml_backend_buffer_get_size(buf.get());
+ buf_host_size = ggml_backend_buffer_get_size(buf_host.get());
+
+ for (group & g : groups) {
+ for (int32_t il : g.layers) {
+ if (layers[il].slot_map != nullptr) {
+ g.lru.slot_map[il] = (int32_t *) layers[il].slot_map->data;
+ }
+ }
+ }
+ }
+
+ // as weights, the ops that read the banks run on the device and the slot maps are copied with the copy callback
+ ggml_backend_buffer_set_usage(buf.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
+ ggml_backend_buffer_set_usage(buf_host.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
+
+ LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__,
+ ggml_backend_buft_name(buft), buf_size/1024.0/1024.0, host_bytes/1024.0/1024.0);
+ for (const group & g : groups) {
+ LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__,
+ g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2]));
+ }
+ }
+
+ ~impl() {
+ log_stats();
+ }
+
+ ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
+ if (il < 0 || il >= (int32_t) layers.size() || layers[il].ig < 0) {
+ return nullptr;
+ }
+ const layer & l = layers[il];
+
+ // large batches use most experts of a layer, so they gain little from the cache and would evict the experts used in generation
+ if (n_tokens == 0 || n_tokens > max_batch || std::min(n_tokens*n_expert_used, l.slot_map->ne[1]) > groups[l.ig].n_slots) {
+ return nullptr;
+ }
+ return l.slot_map;
+ }
+
+ ggml_tensor * get_experts(const ggml_tensor * w) const {
+ const auto it = bindings.find(w);
+ return it != bindings.end() ? it->second.cached : nullptr;
+ }
+
+ int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
+ const auto it = bindings.find(w);
+ if (it == bindings.end() || backend != this->backend) {
+ return 0;
+ }
+ const binding & b = it->second;
+ const group & g = groups[layers[b.il].ig];
+
+ // large batches only read the cache, so the experts used in generation stay in it
+ const int32_t * slots = g.lru.slot_map[b.il];
+ if (slots == nullptr || slots[e] < 0) {
+ return 0;
+ }
+ int64_t n = 1;
+ while (e + n <= last && slots[e + n] == slots[e] + n) {
+ n++;
+ }
+
+ ggml_tensor * bank = g.banks[b.ip];
+ ggml_reset(ctx_views.get());
+ ggml_tensor * src_view = ggml_view_3d(ctx_views.get(), bank, bank->ne[0], bank->ne[1], n, bank->nb[1], bank->nb[2], slots[e]*bank->nb[2]);
+ ggml_tensor * dst_view = ggml_view_3d(ctx_views.get(), dst, dst->ne[0], dst->ne[1], n, dst->nb[1], dst->nb[2], e*dst->nb[2]);
+ ggml_backend_view_init(src_view);
+ ggml_backend_view_init(dst_view);
+ ggml_backend_tensor_copy_async(backend, backend, src_view, dst_view);
+
+ stats_copy.hits += n;
+ stats_copy.bytes += ggml_nbytes(src_view);
+
+ return n;
+ }
+
+ bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) {
+ const auto it = layer_of.find(src);
+ if (it == layer_of.end()) {
+ return false;
+ }
+ const int32_t il = it->second;
+ const layer & l = layers[il];
+ group & g = groups[l.ig];
+
+ GGML_ASSERT(backend == this->backend);
+
+ // the get_rows that looks up the slots of the selected experts
+ const int n_nodes = ggml_graph_n_nodes(graph);
+ const ggml_tensor * lookup = nullptr;
+ for (int i = 0; i < n_nodes && lookup == nullptr; ++i) {
+ const ggml_tensor * node = ggml_graph_node(graph, i);
+ if (node->op == GGML_OP_GET_ROWS && node->src[0] == dst) {
+ lookup = node;
+ }
+ }
+ GGML_ASSERT(lookup != nullptr);
+
+ // the selected experts must be computed in an earlier split
+ // the scheduler starts a new split at the lookup because it reads a host weight, but only if the split already has inputs
+ const ggml_tensor * sel = lookup->src[1];
+ for (int i = 0; i < n_nodes; ++i) {
+ const ggml_tensor * node = ggml_graph_node(graph, i);
+ if (node == sel || node == sel->view_src) {
+ GGML_ABORT("the experts of layer %d are selected in the same split as their MoE cache lookup", il);
+ }
+ }
+ GGML_ASSERT(ggml_is_contiguous(sel));
+
+ ids.resize(ggml_nelements(sel));
+ ggml_backend_tensor_get_async(backend, sel, ids.data(), 0, ggml_nbytes(sel));
+ ggml_backend_synchronize(backend);
+
+ size_t n_hit = 0;
+ if (!g.lru.plan(il, ids.data(), ids.size(), fills, n_hit)) {
+ GGML_ABORT("the MoE cache is too small for the experts selected in layer %d", il);
+ }
+
+ // upload the missing experts, consecutive experts going to consecutive slots are uploaded together
+ size_t bytes = 0;
+ for (size_t ip = 0; ip < l.experts.size(); ++ip) {
+ const ggml_tensor * w = l.experts[ip];
+ ggml_tensor * bank = g.banks[ip];
+ const size_t expert_size = w->nb[2];
+ for (size_t i = 0; i < fills.size();) {
+ size_t n = 1;
+ while (i + n < fills.size() && fills[i + n].expert == fills[i].expert + (int32_t) n && fills[i + n].slot == fills[i].slot + (int32_t) n) {
+ n++;
+ }
+ ggml_backend_tensor_set_async(backend, bank, (const uint8_t *) w->data + fills[i].expert*expert_size, fills[i].slot*expert_size, n*expert_size);
+ bytes += n*expert_size;
+ i += n;
+ }
+ }
+
+ stats & st = ids.size() <= (size_t) 8*n_expert_used ? stats_small : stats_large;
+ st.hits += n_hit;
+ st.misses += fills.size();
+ st.bytes += bytes;
+
+ // the next copy synchronizes the backend before it changes the slot map again
+ ggml_backend_tensor_set_async(backend, dst, src->data, 0, ggml_nbytes(src));
+
+ return true;
+ }
+
+ void log_stats() const {
+ auto log = [](const char * name, const stats & st) {
+ const size_t n = st.hits + st.misses;
+ if (n == 0) {
+ return;
+ }
+ LLAMA_LOG_INFO("llama_moe_cache: %s: hits = %zu, misses = %zu, hit rate = %.2f%%, uploaded = %.2f MiB\n",
+ name, st.hits, st.misses, 100.0*st.hits/n, st.bytes/1024.0/1024.0);
+ };
+ log("ubatch <= 8", stats_small);
+ log("ubatch > 8", stats_large);
+ if (stats_copy.hits > 0) {
+ LLAMA_LOG_INFO("llama_moe_cache: large batches: %zu experts copied from the cache, %.2f MiB\n", stats_copy.hits, stats_copy.bytes/1024.0/1024.0);
+ }
+ }
+};
+
+llama_moe_cache::llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
+ pimpl(new impl(model, backend, buft, size)) {
+}
+
+llama_moe_cache::~llama_moe_cache() = default;
+
+ggml_backend_t llama_moe_cache::backend() const {
+ return pimpl->backend;
+}
+
+ggml_tensor * llama_moe_cache::get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
+ return pimpl->get_slot_map(il, n_tokens, n_expert_used);
+}
+
+ggml_tensor * llama_moe_cache::get_experts(const ggml_tensor * w) const {
+ return pimpl->get_experts(w);
+}
+
+bool llama_moe_cache::copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) {
+ return pimpl->copy(backend, src, dst, graph);
+}
+
+int64_t llama_moe_cache::copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
+ return pimpl->copy_experts(backend, w, dst, e, last);
+}
+
+std::map<ggml_backend_buffer_type_t, size_t> llama_moe_cache::memory_breakdown() const {
+ std::map<ggml_backend_buffer_type_t, size_t> res;
+ if (pimpl->buf) {
+ res[ggml_backend_buffer_get_type(pimpl->buf.get())] += pimpl->buf_size;
+ }
+ if (pimpl->buf_host) {
+ res[ggml_backend_buffer_get_type(pimpl->buf_host.get())] += pimpl->buf_host_size;
+ }
+ return res;
+}
diff --git a/src/llama-moe-cache.h b/src/llama-moe-cache.h
new file mode 100644
index 000000000..4872370d4
--- /dev/null
+++ b/src/llama-moe-cache.h
@@ -0,0 +1,39 @@
+#pragma once
+
+#include "ggml-backend.h"
+
+#include <map>
+#include <memory>
+
+struct llama_model;
+
+// keeps the most recently used experts of host-resident MoE layers in a device buffer
+// each layer has a slot map in host memory: when the scheduler copies it to the device, the copy callback uploads the missing experts
+class llama_moe_cache {
+public:
+ llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size);
+ ~llama_moe_cache();
+
+ ggml_backend_t backend() const;
+
+ // the slot map of layer il, if its experts can be read from the cache for n_tokens tokens, nullptr otherwise
+ ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const;
+
+ // the experts of w in the cache, nullptr if w is not cached
+ ggml_tensor * get_experts(const ggml_tensor * w) const;
+
+ // ggml_backend_sched copy callback, returns false if src is not a slot map
+ bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph);
+
+ // for large batches: copy the experts of w that are in the cache, starting at expert e and up to expert last, to the copy dst of w
+ // returns the number of experts copied, 0 if expert e is not in the cache
+ int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last);
+
+ std::map<ggml_backend_buffer_type_t, size_t> memory_breakdown() const;
+
+private:
+ struct impl;
+ std::unique_ptr<impl> pimpl;
+};
+
+using llama_moe_cache_ptr = std::unique_ptr<llama_moe_cache>;
diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp
index 54122629a..b54b3cc18 100644
--- a/tests/test-llama-archs.cpp
+++ b/tests/test-llama-archs.cpp
@@ -495,7 +495,7 @@ static std::pair<llama_model_ptr, llama_context_ptr> get_model_and_ctx(
struct gguf_context * gguf_ctx, FILE * file, const size_t seed, const float stdev,
const std::vector<ggml_backend_dev_t> & devs,
const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false,
- const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr) {
+ const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr, const size_t moe_cache_size = 0) {
GGML_ASSERT((gguf_ctx == nullptr) != (file == nullptr));
llama_model_params model_params = llama_model_default_params();
model_params.progress_callback = silent_model_load_progress;
@@ -512,6 +512,11 @@ static std::pair<llama_model_ptr, llama_context_ptr> get_model_and_ctx(
if (!encode) {
ctx_params.n_ubatch = 64;
}
+ if (moe_cache_size > 0) {
+ // the MoE cache is only used for small ubatches
+ ctx_params.moe_cache_size = moe_cache_size;
+ ctx_params.n_ubatch = 2;
+ }
tensor_data_params tensor_params = { seed, stdev };
llama_model_ptr model(gguf_ctx != nullptr ?
@@ -865,9 +870,10 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
std::string label;
llama_split_mode split_mode;
bool host_experts; // keep the experts in host memory, see host_experts_test
+ size_t moe_cache_size;
- device_config(std::vector<ggml_backend_dev_t> devs, std::string name, llama_split_mode split_mode, bool host_experts = false)
- : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts) {}
+ device_config(std::vector<ggml_backend_dev_t> devs, std::string name, llama_split_mode split_mode, bool host_experts = false, size_t moe_cache_size = 0)
+ : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts), moe_cache_size(moe_cache_size) {}
};
const llama_model_tensor_buft_override host_experts_overrides[] = {
@@ -905,6 +911,16 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
dev_configs.emplace_back(devices_meta, "Host experts", LLAMA_SPLIT_MODE_LAYER, true);
max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
}
+
+ // the ops that use the host experts run on a GPU and read the experts from a cache
+ // the cache has only a few slots (4 for 288 KiB experts), so the experts are evicted and uploaded again
+ if (!devices_meta.empty()) {
+ const enum ggml_backend_dev_type type = ggml_backend_dev_type(devices_meta[0]);
+ if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
+ dev_configs.emplace_back(std::vector<ggml_backend_dev_t>{devices_meta[0]}, "MoE cache", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024);
+ max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
+ }
+ }
}
size_t max_arch_name_length = 0;
@@ -987,7 +1003,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
}
if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) {
test_executed = true;
- model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides);
+ model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size);
logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode);
const double nmse_val = nmse(logits_cpu, logits_dev);
snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val);
@@ -1053,7 +1069,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
ms.save(file);
rewind(file);
- auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides);
+ auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size);
const std::vector<float> logits_roundtrip = get_logits(
model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode);
status_roundtrip = "\033[1;32mOK\033[0m";