llama : add a GPU cache for MoE experts kept in host memory (#29887)

* llama : add a GPU cache for MoE experts kept in host memory

Assisted-by: Claude

* use llama_moe_cache_ptr
This commit is contained in:
Aman Gupta
2026-10-07 21:07:45 +03:00
committed by GitHub
parent 42c787e8c1
commit d6cf9acb25
14 changed files with 761 additions and 18 deletions
+10
View File
@@ -2776,6 +2776,16 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
llm_add_n_cpu_ffn_overrides(value, LLM_FFN_EXPS_REGEX, params.tensor_buft_overrides); llm_add_n_cpu_ffn_overrides(value, LLM_FFN_EXPS_REGEX, params.tensor_buft_overrides);
} }
).set_env("LLAMA_ARG_N_CPU_MOE")); ).set_env("LLAMA_ARG_N_CPU_MOE"));
add_opt(common_arg(
{"--moe-cache-mib"}, "N",
"GPU cache size in MiB for the MoE experts kept in the CPU (default: 0, disabled)",
[](common_params & params, int value) {
if (value < 0) {
throw std::invalid_argument("invalid value");
}
params.moe_cache_size = (size_t) value*1024*1024;
}
).set_env("LLAMA_ARG_MOE_CACHE_MIB"));
add_opt(common_arg( add_opt(common_arg(
{"-ncffn", "--n-cpu-ffn"}, "N", {"-ncffn", "--n-cpu-ffn"}, "N",
"keep the dense FFN weights of the first N layers in the CPU\n" "keep the dense FFN weights of the first N layers in the CPU\n"
+2
View File
@@ -1722,6 +1722,8 @@ struct llama_context_params common_context_params_to_llama(const common_params &
cparams.type_k = params.cache_type_k; cparams.type_k = params.cache_type_k;
cparams.type_v = params.cache_type_v; cparams.type_v = params.cache_type_v;
cparams.moe_cache_size = params.moe_cache_size;
return cparams; return cparams;
} }
+2
View File
@@ -593,6 +593,8 @@ struct common_params {
ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K
ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V
size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU
common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO; common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;
// multimodal models (see tools/mtmd) // multimodal models (see tools/mtmd)
+3
View File
@@ -2561,6 +2561,9 @@ common_params common_base_params_to_speculative(const common_params & params) {
result.n_outputs_max = params.n_parallel; result.n_outputs_max = params.n_parallel;
result.n_outputs_max_per_seq = 1; result.n_outputs_max_per_seq = 1;
// the MoE cache is only used by the target context
result.moe_cache_size = 0;
// dflash/dspark decode the whole noise block in a single pass and sample every block position on the backend // dflash/dspark decode the whole noise block in a single pass and sample every block position on the backend
// TODO: refactor such properties to be announced by the speculative types // TODO: refactor such properties to be announced by the speculative types
// something like `struct common_speculative_type_props common_speculative_type_get_props(...);` // something like `struct common_speculative_type_props common_speculative_type_get_props(...);`
+2
View File
@@ -396,6 +396,8 @@ extern "C" {
enum ggml_type type_k; // data type for K cache [EXPERIMENTAL] enum ggml_type type_k; // data type for K cache [EXPERIMENTAL]
enum ggml_type type_v; // data type for V cache [EXPERIMENTAL] enum ggml_type type_v; // data type for V cache [EXPERIMENTAL]
size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, 0 = disabled [EXPERIMENTAL]
// Abort callback // Abort callback
// if it returns true, execution of llama_decode() will be aborted // if it returns true, execution of llama_decode() will be aborted
// currently works only with CPU execution // currently works only with CPU execution
+1
View File
@@ -28,6 +28,7 @@ set(LLAMA_CORE_SOURCES
llama-kv-cache-msa.cpp llama-kv-cache-msa.cpp
llama-kv-cache-dsv4.cpp llama-kv-cache-dsv4.cpp
llama-memory.cpp llama-memory.cpp
llama-moe-cache.cpp
llama-memory-hybrid.cpp llama-memory-hybrid.cpp
llama-memory-hybrid-iswa.cpp llama-memory-hybrid-iswa.cpp
llama-memory-hybrid-idx.cpp llama-memory-hybrid-idx.cpp
+55 -5
View File
@@ -9,6 +9,7 @@
#include "llama-memory.h" #include "llama-memory.h"
#include "llama-mmap.h" #include "llama-mmap.h"
#include "llama-model.h" #include "llama-model.h"
#include "llama-moe-cache.h"
#include "llama-ext.h" #include "llama-ext.h"
#include "llama-sampler.h" #include "llama-sampler.h"
#include "llama.h" #include "llama.h"
@@ -272,8 +273,9 @@ llama_context::llama_context(
} }
} }
cparams.op_offload = params.op_offload; cparams.op_offload = params.op_offload;
cparams.kv_unified = params.kv_unified; cparams.kv_unified = params.kv_unified;
cparams.moe_cache_size = params.moe_cache_size;
// initialized later // initialized later
cparams.pipeline_parallel = false; cparams.pipeline_parallel = false;
@@ -462,6 +464,22 @@ llama_context::llama_context(
LLAMA_LOG_INFO("%s: pipeline parallelism enabled\n", __func__); LLAMA_LOG_INFO("%s: pipeline parallelism enabled\n", __func__);
} }
if (cparams.moe_cache_size > 0) {
if (cparams.pipeline_parallel || model.n_devices() > 1) {
throw std::runtime_error("MoE cache does not support multiple devices");
}
for (size_t i = 0; i < backend_ptrs.size(); ++i) {
const auto type = ggml_backend_dev_type(ggml_backend_get_device(backend_ptrs[i]));
if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
moe_cache = std::make_unique<llama_moe_cache>(model, backend_ptrs[i], backend_buft[i], cparams.moe_cache_size);
break;
}
}
if (!moe_cache) {
throw std::runtime_error("MoE cache requires a GPU backend");
}
}
sched_reserve(); sched_reserve();
if (!cparams.flash_attn) { if (!cparams.flash_attn) {
@@ -2605,6 +2623,7 @@ llm_graph_params llama_context::graph_params(
/*.loras =*/ loras.get(), /*.loras =*/ loras.get(),
/*.mctx =*/ mctx, /*.mctx =*/ mctx,
/*.cross =*/ &cross, /*.cross =*/ &cross,
/*.moe_cache =*/ moe_cache.get(),
/*.prec_policy =*/ &model.prec_policy, /*.prec_policy =*/ &model.prec_policy,
/*.samplers =*/ sampling.samplers, /*.samplers =*/ sampling.samplers,
/*.n_outputs =*/ n_outputs, /*.n_outputs =*/ n_outputs,
@@ -2645,7 +2664,14 @@ ggml_status llama_context::graph_compute(
} }
bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data) { bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data) {
auto & st = static_cast<llama_context *>(user_data)->copy_experts; auto * lctx = static_cast<llama_context *>(user_data);
// the slot maps of the MoE cache
if (lctx->moe_cache && lctx->moe_cache->copy(backend, src, dst, graph)) {
return true;
}
auto & st = lctx->copy_experts;
// the ids must be computed before the split starts, so only the first node of the split is considered // the ids must be computed before the split starts, so only the first node of the split is considered
if (ggml_graph_n_nodes(graph) == 0) { if (ggml_graph_n_nodes(graph) == 0) {
@@ -2692,10 +2718,28 @@ bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor
last++; last++;
} }
// the experts in the MoE cache are copied from device memory, the others are uploaded
int64_t next = first;
for (int64_t e = first; e <= last && lctx->moe_cache; ) {
const int64_t n = lctx->moe_cache->copy_experts(backend, src, dst, e, last);
if (n == 0) {
e++;
continue;
}
if (next < e) {
ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + next*expert_size, next*expert_size, (e - next)*expert_size);
}
e += n;
next = e;
}
// copy a bit extra to ensure there are no NaNs in the padding of the last expert, this is necessary for MMQ in the CUDA backend // copy a bit extra to ensure there are no NaNs in the padding of the last expert, this is necessary for MMQ in the CUDA backend
const size_t offset = first*expert_size; const size_t offset = next*expert_size;
const size_t padding = last < n_expert - 1 ? std::min<size_t>(expert_size, 512) : 0; const size_t padding = last < n_expert - 1 ? std::min<size_t>(expert_size, 512) : 0;
ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, (last - first + 1)*expert_size + padding); const size_t size = (last + 1 - next)*expert_size + padding;
if (size > 0) {
ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, size);
}
first = last + 1; first = last + 1;
} }
@@ -3562,6 +3606,11 @@ llama_memory_breakdown llama_context::memory_breakdown() const {
ret[buft].context += size; ret[buft].context += size;
} }
} }
if (moe_cache) {
for (const auto & [buft, size] : moe_cache->memory_breakdown()) {
ret[buft].context += size;
}
}
if (model.hparams.no_alloc) { if (model.hparams.no_alloc) {
for (size_t i = 0; i < backends.size(); ++i) { for (size_t i = 0; i < backends.size(); ++i) {
ggml_backend_t backend = backends[i].get(); ggml_backend_t backend = backends[i].get();
@@ -3851,6 +3900,7 @@ llama_context_params llama_context_default_params() {
/*.cb_eval_user_data =*/ nullptr, /*.cb_eval_user_data =*/ nullptr,
/*.type_k =*/ GGML_TYPE_F16, /*.type_k =*/ GGML_TYPE_F16,
/*.type_v =*/ GGML_TYPE_F16, /*.type_v =*/ GGML_TYPE_F16,
/*.moe_cache_size =*/ 0,
/*.abort_callback =*/ nullptr, /*.abort_callback =*/ nullptr,
/*.abort_callback_data =*/ nullptr, /*.abort_callback_data =*/ nullptr,
/*.embeddings =*/ false, /*.embeddings =*/ false,
+4 -1
View File
@@ -7,6 +7,7 @@
#include "llama-adapter.h" #include "llama-adapter.h"
#include "llama-impl.h" #include "llama-impl.h"
#include "llama-memory.h" #include "llama-memory.h"
#include "llama-moe-cache.h"
#include "ggml-cpp.h" #include "ggml-cpp.h"
#include "ggml-opt.h" #include "ggml-opt.h"
@@ -17,6 +18,7 @@
struct llama_model; struct llama_model;
class llama_batch_allocr; class llama_batch_allocr;
class llama_moe_cache;
class llama_io_read_i; class llama_io_read_i;
class llama_io_write_i; class llama_io_write_i;
@@ -271,7 +273,7 @@ private:
llm_graph_cb graph_get_cb() const; llm_graph_cb graph_get_cb() const;
// ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID // ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID and updates the MoE cache
static bool sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data); static bool sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data);
// disable auto fused ops (Flash Attention, Gated Delta Net) whose op lands on a device // disable auto fused ops (Flash Attention, Gated Delta Net) whose op lands on a device
@@ -299,6 +301,7 @@ private:
llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably
llama_memory_ptr memory; llama_memory_ptr memory;
llama_moe_cache_ptr moe_cache;
// decode output (2-dimensional array: [n_outputs][n_vocab]) // decode output (2-dimensional array: [n_outputs][n_vocab])
buffer_view<float> logits = {nullptr, 0}; buffer_view<float> logits = {nullptr, 0};
+2
View File
@@ -55,6 +55,8 @@ struct llama_cparams {
bool pipeline_parallel; bool pipeline_parallel;
bool training; // set by llama_opt_init() bool training; // set by llama_opt_init()
size_t moe_cache_size;
std::vector<bool> embeddings_layer_inp; // [n_layer()] extract input embeddings for layer std::vector<bool> embeddings_layer_inp; // [n_layer()] extract input embeddings for layer
enum llama_context_type ctx_type; enum llama_context_type ctx_type;
+54 -6
View File
@@ -2,6 +2,7 @@
#include "llama-impl.h" #include "llama-impl.h"
#include "llama-model.h" #include "llama-model.h"
#include "llama-moe-cache.h"
#include "llama-batch.h" #include "llama-batch.h"
#include "llama-cparams.h" #include "llama-cparams.h"
#include "llama-sampler.h" #include "llama-sampler.h"
@@ -1523,6 +1524,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) :
loras (params.loras), loras (params.loras),
mctx (params.mctx), mctx (params.mctx),
cross (params.cross), cross (params.cross),
moe_cache (params.moe_cache),
prec_policy (params.prec_policy), prec_policy (params.prec_policy),
samplers (params.samplers), samplers (params.samplers),
cb_func (params.cb), cb_func (params.cb),
@@ -1589,8 +1591,12 @@ ggml_tensor * llm_graph_context::build_lora_mm_id(
ggml_tensor * w, // ggml_tensor * as ggml_tensor * w, // ggml_tensor * as
ggml_tensor * cur, // ggml_tensor * b ggml_tensor * cur, // ggml_tensor * b
ggml_tensor * ids, ggml_tensor * ids,
ggml_tensor * w_s) const { ggml_tensor * w_s,
ggml_tensor * res = ggml_mul_mat_id(ctx0, w, cur, ids); ggml_tensor * slots) const {
// the experts in the MoE cache are selected by their slots
ggml_tensor * res = slots == nullptr ?
ggml_mul_mat_id(ctx0, w, cur, ids) :
ggml_mul_mat_id(ctx0, moe_cache->get_experts(w), cur, slots);
if (prec_policy) { if (prec_policy) {
prec_policy->apply(res); prec_policy->apply(res);
@@ -2205,6 +2211,9 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
//call early so that topk-moe can be used //call early so that topk-moe can be used
ggml_build_forward_expand(gf, weights); ggml_build_forward_expand(gf, weights);
// the experts of host-resident layers may be read from the MoE cache
ggml_tensor * slots = build_moe_cache_slots(selected_experts, up_exps, gate_exps, down_exps, gate_up_exps, il);
cur = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens); cur = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
if (weight_before_ffn) { if (weight_before_ffn) {
@@ -2219,7 +2228,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
if (gate_up_exps) { if (gate_up_exps) {
// merged gate_up path: one mul_mat_id, then split into gate and up views // merged gate_up path: one mul_mat_id, then split into gate and up views
ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s); // [n_ff*2, n_expert_used, n_tokens] ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff*2, n_expert_used, n_tokens]
cb(gate_up, "ffn_moe_gate_up", il); cb(gate_up, "ffn_moe_gate_up", il);
if (up_exps_s) { if (up_exps_s) {
@@ -2238,7 +2247,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
cb(up, "ffn_moe_up", il); cb(up, "ffn_moe_up", il);
} else { } else {
// separate gate and up path // separate gate and up path
up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s); // [n_ff, n_expert_used, n_tokens] up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff, n_expert_used, n_tokens]
cb(up, "ffn_moe_up", il); cb(up, "ffn_moe_up", il);
if (up_exps_s) { if (up_exps_s) {
@@ -2251,7 +2260,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
} }
if (gate_exps) { if (gate_exps) {
cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s); // [n_ff, n_expert_used, n_tokens] cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s, slots); // [n_ff, n_expert_used, n_tokens]
cb(cur, "ffn_moe_gate", il); cb(cur, "ffn_moe_gate", il);
} else { } else {
cur = up; cur = up;
@@ -2352,7 +2361,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
GGML_ABORT("fatal error"); GGML_ABORT("fatal error");
} }
experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s); // [n_embd, n_expert_used, n_tokens] experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s, slots); // [n_embd, n_expert_used, n_tokens]
if (arch == LLM_ARCH_MISTRAL4) { if (arch == LLM_ARCH_MISTRAL4) {
// src1 can exceed F16 range // src1 can exceed F16 range
ggml_prec_set_src(experts, GGML_PREC_F32, 1); ggml_prec_set_src(experts, GGML_PREC_F32, 1);
@@ -2409,6 +2418,45 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
return moe_out; return moe_out;
} }
ggml_tensor * llm_graph_context::build_moe_cache_slots(
ggml_tensor * selected_experts,
ggml_tensor * up_exps,
ggml_tensor * gate_exps,
ggml_tensor * down_exps,
ggml_tensor * gate_up_exps,
int il) const {
if (moe_cache == nullptr) {
return nullptr;
}
ggml_tensor * slot_map = moe_cache->get_slot_map(il, selected_experts->ne[1], selected_experts->ne[0]);
if (slot_map == nullptr) {
return nullptr;
}
for (ggml_tensor * w : { up_exps, gate_exps, down_exps, gate_up_exps }) {
if (w != nullptr && moe_cache->get_experts(w) == nullptr) {
return nullptr;
}
}
ggml_tensor * ids = selected_experts;
if (!ggml_is_contiguous(ids)) {
ids = ggml_cont(ctx0, ids);
}
ids = ggml_reshape_1d(ctx0, ids, ggml_nelements(ids));
// the slot map is a host weight, so the scheduler starts a new split here and copies it with the copy callback
// the callback reads the selected experts, uploads the missing ones and updates the slot map
ggml_tensor * slots = ggml_get_rows(ctx0, slot_map, ids); // [1, n_expert_used*n_tokens]
if (!ggml_backend_supports_op(moe_cache->backend(), slots)) {
return nullptr;
}
ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend());
cb(slots, "ffn_moe_slots", il);
return ggml_reshape_2d(ctx0, slots, selected_experts->ne[0], selected_experts->ne[1]); // [n_expert_used, n_tokens]
}
// input embeddings with optional lora // input embeddings with optional lora
ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float tok_scale) const { ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float tok_scale) const {
const int64_t n_embd_inp = hparams.n_embd_inp(); const int64_t n_embd_inp = hparams.n_embd_inp();
+16 -1
View File
@@ -21,6 +21,8 @@ struct llama_cparams;
struct llama_layer; struct llama_layer;
struct llama_prec_policy; struct llama_prec_policy;
class llama_moe_cache;
struct llama_memory_context_i; struct llama_memory_context_i;
class llama_kv_cache_context; class llama_kv_cache_context;
@@ -793,6 +795,7 @@ struct llm_graph_params {
const llama_adapter_loras * loras; const llama_adapter_loras * loras;
const llama_memory_context_i * mctx; const llama_memory_context_i * mctx;
const llama_cross * cross; const llama_cross * cross;
const llama_moe_cache * moe_cache;
const llama_prec_policy * prec_policy = nullptr; const llama_prec_policy * prec_policy = nullptr;
@@ -1036,6 +1039,7 @@ struct llm_graph_context {
const llama_adapter_loras * loras; const llama_adapter_loras * loras;
const llama_memory_context_i * mctx; const llama_memory_context_i * mctx;
const llama_cross * cross; const llama_cross * cross;
const llama_moe_cache * moe_cache;
const llama_prec_policy * prec_policy; const llama_prec_policy * prec_policy;
@@ -1078,11 +1082,13 @@ struct llm_graph_context {
ggml_tensor * w_s = nullptr) const; ggml_tensor * w_s = nullptr) const;
// do mat_mul_id, while optionally apply lora and per-expert scale // do mat_mul_id, while optionally apply lora and per-expert scale
// if slots is set, the experts are read from the MoE cache at these slots (see build_moe_cache_slots)
ggml_tensor * build_lora_mm_id( ggml_tensor * build_lora_mm_id(
ggml_tensor * w, // ggml_tensor * as ggml_tensor * w, // ggml_tensor * as
ggml_tensor * cur, // ggml_tensor * b ggml_tensor * cur, // ggml_tensor * b
ggml_tensor * ids, ggml_tensor * ids,
ggml_tensor * w_s = nullptr) const; ggml_tensor * w_s = nullptr,
ggml_tensor * slots = nullptr) const;
ggml_tensor * build_norm( ggml_tensor * build_norm(
ggml_tensor * cur, ggml_tensor * cur,
@@ -1179,6 +1185,15 @@ struct llm_graph_context {
ggml_tensor * down_exps_s = nullptr, ggml_tensor * down_exps_s = nullptr,
ggml_tensor * selected_experts_in = nullptr) const; ggml_tensor * selected_experts_in = nullptr) const;
// the slots of the selected experts in the MoE cache, nullptr if the experts of the layer are not read from the cache
ggml_tensor * build_moe_cache_slots(
ggml_tensor * selected_experts,
ggml_tensor * up_exps,
ggml_tensor * gate_exps,
ggml_tensor * down_exps,
ggml_tensor * gate_up_exps,
int il) const;
// //
// inputs // inputs
// //
+550
View File
@@ -0,0 +1,550 @@
#include "llama-moe-cache.h"
#include "llama-impl.h"
#include "llama-model.h"
#include "ggml-cpp.h"
#include <algorithm>
#include <stdexcept>
#include <unordered_map>
#include <vector>
namespace {
// LRU of the experts of a group of layers, the slot of each expert is kept in the slot map of its layer
struct moe_cache_lru {
int32_t n_expert = 0;
int32_t n_slots = 0;
std::vector<int32_t *> slot_map; // [n_layer] data of the slot maps, -1 if the expert is not cached
std::vector<int32_t> key_of; // [n_slots] il*n_expert + expert, -1 if empty
// doubly linked list of the slots, head is the least recently used
std::vector<int32_t> prev;
std::vector<int32_t> next;
int32_t head = -1;
int32_t tail = -1;
std::vector<uint32_t> seen; // [n_expert]
uint32_t seen_gen = 0;
std::vector<int32_t> uniq;
void init(int32_t n_layer, int32_t n_expert, int32_t n_slots) {
this->n_expert = n_expert;
this->n_slots = n_slots;
slot_map.assign(n_layer, nullptr);
key_of.assign(n_slots, -1);
prev.resize(n_slots);
next.resize(n_slots);
for (int32_t s = 0; s < n_slots; ++s) {
prev[s] = s - 1;
next[s] = s + 1 < n_slots ? s + 1 : -1;
}
head = 0;
tail = n_slots - 1;
seen.assign(n_expert, 0);
}
// move slot s to the tail (most recently used)
void touch(int32_t s) {
if (s == tail) {
return;
}
if (prev[s] >= 0) {
next[prev[s]] = next[s];
} else {
head = next[s];
}
prev[next[s]] = prev[s];
prev[s] = tail;
next[s] = -1;
next[tail] = s;
tail = s;
}
struct fill {
int32_t expert;
int32_t slot;
};
// give a slot to each expert selected by ids in layer il, the misses evict the least recently used experts
// returns false if the ids select more distinct experts than there are slots
bool plan(int32_t il, const int32_t * ids, size_t n_ids, std::vector<fill> & fills, size_t & n_hit) {
fills.clear();
n_hit = 0;
if (++seen_gen == 0) {
std::fill(seen.begin(), seen.end(), 0);
seen_gen = 1;
}
uniq.clear();
for (size_t i = 0; i < n_ids; ++i) {
GGML_ASSERT(ids[i] >= 0 && ids[i] < n_expert);
if (seen[ids[i]] != seen_gen) {
seen[ids[i]] = seen_gen;
uniq.push_back(ids[i]);
}
}
if (uniq.size() > (size_t) n_slots) {
return false;
}
int32_t * slots = slot_map[il];
// hits go to the tail first, so the head can be evicted below
for (int32_t e : uniq) {
if (slots[e] >= 0) {
touch(slots[e]);
n_hit++;
}
}
// sorted misses usually get consecutive slots, so the uploads can be merged
std::sort(uniq.begin(), uniq.end());
for (int32_t e : uniq) {
if (slots[e] >= 0) {
continue;
}
const int32_t s = head;
if (key_of[s] >= 0) {
slot_map[key_of[s] / n_expert][key_of[s] % n_expert] = -1;
}
key_of[s] = il*n_expert + e;
slots[e] = s;
touch(s);
fills.push_back({ e, s });
}
return true;
}
};
// gate, up, down or gate_up, down
static std::vector<ggml_tensor *> llama_moe_cache_layer_experts(const llama_layer & layer) {
std::vector<ggml_tensor *> res;
for (ggml_tensor * t : { layer.ffn_gate_up_exps, layer.ffn_gate_exps, layer.ffn_up_exps, layer.ffn_down_exps }) {
if (t != nullptr) {
res.push_back(t);
}
}
return res;
}
static bool llama_moe_cache_same_layout(const std::vector<ggml_tensor *> & a, const std::vector<ggml_tensor *> & b) {
if (a.size() != b.size()) {
return false;
}
for (size_t i = 0; i < a.size(); ++i) {
if (a[i]->type != b[i]->type || !ggml_are_same_shape(a[i], b[i]) || a[i]->nb[2] != b[i]->nb[2]) {
return false;
}
}
return true;
}
static bool llama_moe_cache_is_host_weight(const ggml_tensor * t) {
return t->buffer != nullptr &&
ggml_backend_buffer_get_usage(t->buffer) == GGML_BACKEND_BUFFER_USAGE_WEIGHTS &&
ggml_backend_buffer_is_host(t->buffer);
}
}
struct llama_moe_cache::impl {
// layers with the same expert tensor layout share the banks and the LRU of a group
struct group {
std::vector<ggml_tensor *> ref; // expert tensors of the first layer
std::vector<int32_t> layers;
std::vector<ggml_tensor *> banks; // device storage of all slots, one per expert tensor
size_t host_bytes = 0;
int32_t n_slots = 0;
moe_cache_lru lru;
};
struct layer {
int32_t ig = -1; // -1 if the layer is not cached
ggml_tensor * slot_map = nullptr; // I32 [1, n_expert] in host memory
std::vector<ggml_tensor *> experts; // host expert tensors, in the order of the banks
};
struct binding {
int32_t il;
int32_t ip; // index of the bank
ggml_tensor * cached; // view of the bank used in place of the host experts
};
struct stats {
size_t hits = 0;
size_t misses = 0;
size_t bytes = 0;
};
static constexpr int64_t max_batch = 32;
ggml_backend_t backend;
int32_t n_expert_used;
stats stats_small; // up to 8 tokens per ubatch
stats stats_large;
stats stats_copy; // experts copied from the cache for large batches
std::vector<group> groups;
std::vector<layer> layers;
std::unordered_map<const ggml_tensor *, binding> bindings; // host experts -> cached experts
std::unordered_map<const ggml_tensor *, int32_t> layer_of; // slot map -> layer
std::vector<int32_t> ids;
std::vector<moe_cache_lru::fill> fills;
// banks and their views on the device
ggml_context_ptr ctx;
ggml_backend_buffer_ptr buf;
size_t buf_size = 0;
// slot maps in host memory
ggml_context_ptr ctx_host;
ggml_backend_buffer_ptr buf_host;
size_t buf_host_size = 0;
// views used by copy_experts
ggml_context_ptr ctx_views;
impl(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
backend(backend), n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) {
ggml_backend_dev_t dev = ggml_backend_get_device(backend);
const auto dev_type = ggml_backend_dev_type(dev);
if (dev_type != GGML_BACKEND_DEVICE_TYPE_GPU && dev_type != GGML_BACKEND_DEVICE_TYPE_IGPU) {
throw std::runtime_error("MoE cache requires a GPU backend");
}
if (model.split_mode() == LLAMA_SPLIT_MODE_TENSOR) {
throw std::runtime_error("MoE cache does not support tensor parallelism");
}
if (model.hparams.n_expert == 0 || n_expert_used == 0) {
throw std::runtime_error("MoE cache requires a MoE model");
}
// only cache layers that keep all of their experts in host memory
size_t host_bytes = 0;
for (size_t il = 0; il < model.layers.size(); ++il) {
auto experts = llama_moe_cache_layer_experts(model.layers[il]);
if (experts.empty() || model.dev_layer(il) != dev ||
!std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) {
continue;
}
auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return llama_moe_cache_same_layout(g.ref, experts); });
if (it == groups.end()) {
groups.emplace_back();
it = groups.end() - 1;
it->ref = experts;
}
it->layers.push_back(il);
for (const ggml_tensor * t : experts) {
it->host_bytes += ggml_nbytes(t);
host_bytes += ggml_nbytes(t);
}
}
if (groups.empty()) {
LLAMA_LOG_WARN("%s: no layer has all of its experts in host memory, MoE cache is disabled\n", __func__);
return;
}
// one extra slot at the end, CUDA MMQ can read past the last expert
const size_t alignment = ggml_backend_buft_get_alignment(buft);
auto alloc_size = [&](const group & g, int32_t n_slots) {
size_t res = 0;
for (const ggml_tensor * t : g.ref) {
res += GGML_PAD(t->nb[2]*(n_slots + 1), alignment);
}
return res;
};
// split the budget by the size of the experts, so each group caches the same fraction of its experts
size_t n_tensors = 0;
size_t n_tensors_host = 0;
for (group & g : groups) {
const int32_t n_expert = g.ref[0]->ne[2];
const size_t budget = (size_t) ((double) size*g.host_bytes/host_bytes);
const int32_t max_slots = g.layers.size()*n_expert;
while (g.n_slots < max_slots && alloc_size(g, g.n_slots + 1) <= budget) {
g.n_slots++;
}
if (g.n_slots < n_expert_used) {
LLAMA_LOG_WARN("%s: MoE cache budget is too small for %zu layers, they are not cached\n", __func__, g.layers.size());
g.n_slots = 0;
continue;
}
g.lru.init(model.layers.size(), n_expert, g.n_slots);
n_tensors += g.ref.size()*(1 + g.layers.size());
n_tensors_host += g.layers.size();
}
if (n_tensors == 0) {
throw std::runtime_error("MoE cache is too small to hold the experts of one token");
}
auto init_ctx = [](size_t n_tensors) {
ggml_init_params params = {
/*.mem_size =*/ n_tensors*ggml_tensor_overhead(),
/*.mem_buffer =*/ nullptr,
/*.no_alloc =*/ true,
};
ggml_context_ptr res(ggml_init(params));
if (!res) {
throw std::runtime_error("failed to create the MoE cache context");
}
return res;
};
ctx = init_ctx(n_tensors);
ctx_host = init_ctx(n_tensors_host);
ctx_views = init_ctx(2);
ggml_backend_buffer_type_t buft_host = ggml_backend_cpu_buffer_type();
const size_t alignment_host = ggml_backend_buft_get_alignment(buft_host);
for (size_t ig = 0; ig < groups.size(); ++ig) {
group & g = groups[ig];
if (g.n_slots == 0) {
continue;
}
for (const ggml_tensor * t : g.ref) {
ggml_tensor * bank = ggml_new_tensor_3d(ctx.get(), t->type, t->ne[0], t->ne[1], g.n_slots + 1);
GGML_ASSERT(bank->nb[2] == t->nb[2]);
ggml_format_name(bank, "moe_cache.%zu.%s", ig, t->name);
g.banks.push_back(bank);
}
for (int32_t il : g.layers) {
layer & l = layers[il];
l.ig = (int32_t) ig;
l.experts = llama_moe_cache_layer_experts(model.layers[il]);
for (size_t ip = 0; ip < l.experts.size(); ++ip) {
ggml_tensor * bank = g.banks[ip];
ggml_tensor * cached = ggml_view_3d(ctx.get(), bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0);
ggml_format_name(cached, "moe_cache.%s", l.experts[ip]->name);
bindings[l.experts[ip]] = { il, (int32_t) ip, cached };
}
l.slot_map = ggml_new_tensor_2d(ctx_host.get(), GGML_TYPE_I32, 1, g.ref[0]->ne[2]);
ggml_format_name(l.slot_map, "moe_cache.slot_map-%d", il);
layer_of[l.slot_map] = il;
buf_host_size += GGML_PAD(ggml_nbytes(l.slot_map), alignment_host);
}
buf_size += alloc_size(g, g.n_slots);
}
if (model.hparams.no_alloc) {
// only used to measure the memory use, see llama_context::memory_breakdown
buf.reset(ggml_backend_buft_alloc_buffer(buft, 0));
buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0));
for (ggml_tensor * t = ggml_get_first_tensor(ctx.get()); t != nullptr; t = ggml_get_next_tensor(ctx.get(), t)) {
t->buffer = buf.get();
}
for (ggml_tensor * t = ggml_get_first_tensor(ctx_host.get()); t != nullptr; t = ggml_get_next_tensor(ctx_host.get(), t)) {
t->buffer = buf_host.get();
}
} else {
buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx.get(), buft));
buf_host.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx_host.get(), buft_host));
if (!buf || !buf_host) {
throw std::runtime_error("failed to allocate the MoE cache buffers");
}
ggml_backend_buffer_clear(buf.get(), 0);
ggml_backend_buffer_clear(buf_host.get(), 0xff); // all slots are -1
buf_size = ggml_backend_buffer_get_size(buf.get());
buf_host_size = ggml_backend_buffer_get_size(buf_host.get());
for (group & g : groups) {
for (int32_t il : g.layers) {
if (layers[il].slot_map != nullptr) {
g.lru.slot_map[il] = (int32_t *) layers[il].slot_map->data;
}
}
}
}
// as weights, the ops that read the banks run on the device and the slot maps are copied with the copy callback
ggml_backend_buffer_set_usage(buf.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
ggml_backend_buffer_set_usage(buf_host.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__,
ggml_backend_buft_name(buft), buf_size/1024.0/1024.0, host_bytes/1024.0/1024.0);
for (const group & g : groups) {
LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__,
g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2]));
}
}
~impl() {
log_stats();
}
ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
if (il < 0 || il >= (int32_t) layers.size() || layers[il].ig < 0) {
return nullptr;
}
const layer & l = layers[il];
// large batches use most experts of a layer, so they gain little from the cache and would evict the experts used in generation
if (n_tokens == 0 || n_tokens > max_batch || std::min(n_tokens*n_expert_used, l.slot_map->ne[1]) > groups[l.ig].n_slots) {
return nullptr;
}
return l.slot_map;
}
ggml_tensor * get_experts(const ggml_tensor * w) const {
const auto it = bindings.find(w);
return it != bindings.end() ? it->second.cached : nullptr;
}
int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
const auto it = bindings.find(w);
if (it == bindings.end() || backend != this->backend) {
return 0;
}
const binding & b = it->second;
const group & g = groups[layers[b.il].ig];
// large batches only read the cache, so the experts used in generation stay in it
const int32_t * slots = g.lru.slot_map[b.il];
if (slots == nullptr || slots[e] < 0) {
return 0;
}
int64_t n = 1;
while (e + n <= last && slots[e + n] == slots[e] + n) {
n++;
}
ggml_tensor * bank = g.banks[b.ip];
ggml_reset(ctx_views.get());
ggml_tensor * src_view = ggml_view_3d(ctx_views.get(), bank, bank->ne[0], bank->ne[1], n, bank->nb[1], bank->nb[2], slots[e]*bank->nb[2]);
ggml_tensor * dst_view = ggml_view_3d(ctx_views.get(), dst, dst->ne[0], dst->ne[1], n, dst->nb[1], dst->nb[2], e*dst->nb[2]);
ggml_backend_view_init(src_view);
ggml_backend_view_init(dst_view);
ggml_backend_tensor_copy_async(backend, backend, src_view, dst_view);
stats_copy.hits += n;
stats_copy.bytes += ggml_nbytes(src_view);
return n;
}
bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) {
const auto it = layer_of.find(src);
if (it == layer_of.end()) {
return false;
}
const int32_t il = it->second;
const layer & l = layers[il];
group & g = groups[l.ig];
GGML_ASSERT(backend == this->backend);
// the get_rows that looks up the slots of the selected experts
const int n_nodes = ggml_graph_n_nodes(graph);
const ggml_tensor * lookup = nullptr;
for (int i = 0; i < n_nodes && lookup == nullptr; ++i) {
const ggml_tensor * node = ggml_graph_node(graph, i);
if (node->op == GGML_OP_GET_ROWS && node->src[0] == dst) {
lookup = node;
}
}
GGML_ASSERT(lookup != nullptr);
// the selected experts must be computed in an earlier split
// the scheduler starts a new split at the lookup because it reads a host weight, but only if the split already has inputs
const ggml_tensor * sel = lookup->src[1];
for (int i = 0; i < n_nodes; ++i) {
const ggml_tensor * node = ggml_graph_node(graph, i);
if (node == sel || node == sel->view_src) {
GGML_ABORT("the experts of layer %d are selected in the same split as their MoE cache lookup", il);
}
}
GGML_ASSERT(ggml_is_contiguous(sel));
ids.resize(ggml_nelements(sel));
ggml_backend_tensor_get_async(backend, sel, ids.data(), 0, ggml_nbytes(sel));
ggml_backend_synchronize(backend);
size_t n_hit = 0;
if (!g.lru.plan(il, ids.data(), ids.size(), fills, n_hit)) {
GGML_ABORT("the MoE cache is too small for the experts selected in layer %d", il);
}
// upload the missing experts, consecutive experts going to consecutive slots are uploaded together
size_t bytes = 0;
for (size_t ip = 0; ip < l.experts.size(); ++ip) {
const ggml_tensor * w = l.experts[ip];
ggml_tensor * bank = g.banks[ip];
const size_t expert_size = w->nb[2];
for (size_t i = 0; i < fills.size();) {
size_t n = 1;
while (i + n < fills.size() && fills[i + n].expert == fills[i].expert + (int32_t) n && fills[i + n].slot == fills[i].slot + (int32_t) n) {
n++;
}
ggml_backend_tensor_set_async(backend, bank, (const uint8_t *) w->data + fills[i].expert*expert_size, fills[i].slot*expert_size, n*expert_size);
bytes += n*expert_size;
i += n;
}
}
stats & st = ids.size() <= (size_t) 8*n_expert_used ? stats_small : stats_large;
st.hits += n_hit;
st.misses += fills.size();
st.bytes += bytes;
// the next copy synchronizes the backend before it changes the slot map again
ggml_backend_tensor_set_async(backend, dst, src->data, 0, ggml_nbytes(src));
return true;
}
void log_stats() const {
auto log = [](const char * name, const stats & st) {
const size_t n = st.hits + st.misses;
if (n == 0) {
return;
}
LLAMA_LOG_INFO("llama_moe_cache: %s: hits = %zu, misses = %zu, hit rate = %.2f%%, uploaded = %.2f MiB\n",
name, st.hits, st.misses, 100.0*st.hits/n, st.bytes/1024.0/1024.0);
};
log("ubatch <= 8", stats_small);
log("ubatch > 8", stats_large);
if (stats_copy.hits > 0) {
LLAMA_LOG_INFO("llama_moe_cache: large batches: %zu experts copied from the cache, %.2f MiB\n", stats_copy.hits, stats_copy.bytes/1024.0/1024.0);
}
}
};
llama_moe_cache::llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
pimpl(new impl(model, backend, buft, size)) {
}
llama_moe_cache::~llama_moe_cache() = default;
ggml_backend_t llama_moe_cache::backend() const {
return pimpl->backend;
}
ggml_tensor * llama_moe_cache::get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
return pimpl->get_slot_map(il, n_tokens, n_expert_used);
}
ggml_tensor * llama_moe_cache::get_experts(const ggml_tensor * w) const {
return pimpl->get_experts(w);
}
bool llama_moe_cache::copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) {
return pimpl->copy(backend, src, dst, graph);
}
int64_t llama_moe_cache::copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
return pimpl->copy_experts(backend, w, dst, e, last);
}
std::map<ggml_backend_buffer_type_t, size_t> llama_moe_cache::memory_breakdown() const {
std::map<ggml_backend_buffer_type_t, size_t> res;
if (pimpl->buf) {
res[ggml_backend_buffer_get_type(pimpl->buf.get())] += pimpl->buf_size;
}
if (pimpl->buf_host) {
res[ggml_backend_buffer_get_type(pimpl->buf_host.get())] += pimpl->buf_host_size;
}
return res;
}
+39
View File
@@ -0,0 +1,39 @@
#pragma once
#include "ggml-backend.h"
#include <map>
#include <memory>
struct llama_model;
// keeps the most recently used experts of host-resident MoE layers in a device buffer
// each layer has a slot map in host memory: when the scheduler copies it to the device, the copy callback uploads the missing experts
class llama_moe_cache {
public:
llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size);
~llama_moe_cache();
ggml_backend_t backend() const;
// the slot map of layer il, if its experts can be read from the cache for n_tokens tokens, nullptr otherwise
ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const;
// the experts of w in the cache, nullptr if w is not cached
ggml_tensor * get_experts(const ggml_tensor * w) const;
// ggml_backend_sched copy callback, returns false if src is not a slot map
bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph);
// for large batches: copy the experts of w that are in the cache, starting at expert e and up to expert last, to the copy dst of w
// returns the number of experts copied, 0 if expert e is not in the cache
int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last);
std::map<ggml_backend_buffer_type_t, size_t> memory_breakdown() const;
private:
struct impl;
std::unique_ptr<impl> pimpl;
};
using llama_moe_cache_ptr = std::unique_ptr<llama_moe_cache>;
+21 -5
View File
@@ -495,7 +495,7 @@ static std::pair<llama_model_ptr, llama_context_ptr> get_model_and_ctx(
struct gguf_context * gguf_ctx, FILE * file, const size_t seed, const float stdev, struct gguf_context * gguf_ctx, FILE * file, const size_t seed, const float stdev,
const std::vector<ggml_backend_dev_t> & devs, const std::vector<ggml_backend_dev_t> & devs,
const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false, const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false,
const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr) { const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr, const size_t moe_cache_size = 0) {
GGML_ASSERT((gguf_ctx == nullptr) != (file == nullptr)); GGML_ASSERT((gguf_ctx == nullptr) != (file == nullptr));
llama_model_params model_params = llama_model_default_params(); llama_model_params model_params = llama_model_default_params();
model_params.progress_callback = silent_model_load_progress; model_params.progress_callback = silent_model_load_progress;
@@ -512,6 +512,11 @@ static std::pair<llama_model_ptr, llama_context_ptr> get_model_and_ctx(
if (!encode) { if (!encode) {
ctx_params.n_ubatch = 64; ctx_params.n_ubatch = 64;
} }
if (moe_cache_size > 0) {
// the MoE cache is only used for small ubatches
ctx_params.moe_cache_size = moe_cache_size;
ctx_params.n_ubatch = 2;
}
tensor_data_params tensor_params = { seed, stdev }; tensor_data_params tensor_params = { seed, stdev };
llama_model_ptr model(gguf_ctx != nullptr ? llama_model_ptr model(gguf_ctx != nullptr ?
@@ -865,9 +870,10 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
std::string label; std::string label;
llama_split_mode split_mode; llama_split_mode split_mode;
bool host_experts; // keep the experts in host memory, see host_experts_test bool host_experts; // keep the experts in host memory, see host_experts_test
size_t moe_cache_size;
device_config(std::vector<ggml_backend_dev_t> devs, std::string name, llama_split_mode split_mode, bool host_experts = false) device_config(std::vector<ggml_backend_dev_t> devs, std::string name, llama_split_mode split_mode, bool host_experts = false, size_t moe_cache_size = 0)
: devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts) {} : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts), moe_cache_size(moe_cache_size) {}
}; };
const llama_model_tensor_buft_override host_experts_overrides[] = { const llama_model_tensor_buft_override host_experts_overrides[] = {
@@ -905,6 +911,16 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
dev_configs.emplace_back(devices_meta, "Host experts", LLAMA_SPLIT_MODE_LAYER, true); dev_configs.emplace_back(devices_meta, "Host experts", LLAMA_SPLIT_MODE_LAYER, true);
max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length()); max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
} }
// the ops that use the host experts run on a GPU and read the experts from a cache
// the cache has only a few slots (4 for 288 KiB experts), so the experts are evicted and uploaded again
if (!devices_meta.empty()) {
const enum ggml_backend_dev_type type = ggml_backend_dev_type(devices_meta[0]);
if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
dev_configs.emplace_back(std::vector<ggml_backend_dev_t>{devices_meta[0]}, "MoE cache", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024);
max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
}
}
} }
size_t max_arch_name_length = 0; size_t max_arch_name_length = 0;
@@ -987,7 +1003,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
} }
if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) { if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) {
test_executed = true; test_executed = true;
model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides); model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size);
logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode); logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode);
const double nmse_val = nmse(logits_cpu, logits_dev); const double nmse_val = nmse(logits_cpu, logits_dev);
snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val); snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val);
@@ -1053,7 +1069,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
ms.save(file); ms.save(file);
rewind(file); rewind(file);
auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides); auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size);
const std::vector<float> logits_roundtrip = get_logits( const std::vector<float> logits_roundtrip = get_logits(
model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode); model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode);
status_roundtrip = "\033[1;32mOK\033[0m"; status_roundtrip = "\033[1;32mOK\033[0m";