diff --git a/common/arg.cpp b/common/arg.cpp index 14d82f695c..182f76f45a 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -2776,6 +2776,16 @@ common_params_context common_params_parser_init(common_params & params, llama_ex llm_add_n_cpu_ffn_overrides(value, LLM_FFN_EXPS_REGEX, params.tensor_buft_overrides); } ).set_env("LLAMA_ARG_N_CPU_MOE")); + add_opt(common_arg( + {"--moe-cache-mib"}, "N", + "GPU cache size in MiB for the MoE experts kept in the CPU (default: 0, disabled)", + [](common_params & params, int value) { + if (value < 0) { + throw std::invalid_argument("invalid value"); + } + params.moe_cache_size = (size_t) value*1024*1024; + } + ).set_env("LLAMA_ARG_MOE_CACHE_MIB")); add_opt(common_arg( {"-ncffn", "--n-cpu-ffn"}, "N", "keep the dense FFN weights of the first N layers in the CPU\n" diff --git a/common/common.cpp b/common/common.cpp index e36f8ab502..768b2e9a5c 100644 --- a/common/common.cpp +++ b/common/common.cpp @@ -1722,6 +1722,8 @@ struct llama_context_params common_context_params_to_llama(const common_params & cparams.type_k = params.cache_type_k; cparams.type_v = params.cache_type_v; + cparams.moe_cache_size = params.moe_cache_size; + return cparams; } diff --git a/common/common.h b/common/common.h index 3e3eff379e..0f912d9018 100644 --- a/common/common.h +++ b/common/common.h @@ -593,6 +593,8 @@ struct common_params { ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V + size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU + common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO; // multimodal models (see tools/mtmd) diff --git a/common/speculative.cpp b/common/speculative.cpp index a7f095a121..d9ddf44d28 100644 --- a/common/speculative.cpp +++ b/common/speculative.cpp @@ -2561,6 +2561,9 @@ common_params common_base_params_to_speculative(const common_params & params) { result.n_outputs_max = params.n_parallel; result.n_outputs_max_per_seq = 1; + // the MoE cache is only used by the target context + result.moe_cache_size = 0; + // dflash/dspark decode the whole noise block in a single pass and sample every block position on the backend // TODO: refactor such properties to be announced by the speculative types // something like `struct common_speculative_type_props common_speculative_type_get_props(...);` diff --git a/include/llama.h b/include/llama.h index 260247e827..77f527d5b6 100644 --- a/include/llama.h +++ b/include/llama.h @@ -396,6 +396,8 @@ extern "C" { enum ggml_type type_k; // data type for K cache [EXPERIMENTAL] enum ggml_type type_v; // data type for V cache [EXPERIMENTAL] + size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, 0 = disabled [EXPERIMENTAL] + // Abort callback // if it returns true, execution of llama_decode() will be aborted // currently works only with CPU execution diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index afdaddc79d..97250c2e88 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -28,6 +28,7 @@ set(LLAMA_CORE_SOURCES llama-kv-cache-msa.cpp llama-kv-cache-dsv4.cpp llama-memory.cpp + llama-moe-cache.cpp llama-memory-hybrid.cpp llama-memory-hybrid-iswa.cpp llama-memory-hybrid-idx.cpp diff --git a/src/llama-context.cpp b/src/llama-context.cpp index ff2ea461c4..5f861f96c8 100644 --- a/src/llama-context.cpp +++ b/src/llama-context.cpp @@ -9,6 +9,7 @@ #include "llama-memory.h" #include "llama-mmap.h" #include "llama-model.h" +#include "llama-moe-cache.h" #include "llama-ext.h" #include "llama-sampler.h" #include "llama.h" @@ -272,8 +273,9 @@ llama_context::llama_context( } } - cparams.op_offload = params.op_offload; - cparams.kv_unified = params.kv_unified; + cparams.op_offload = params.op_offload; + cparams.kv_unified = params.kv_unified; + cparams.moe_cache_size = params.moe_cache_size; // initialized later cparams.pipeline_parallel = false; @@ -462,6 +464,22 @@ llama_context::llama_context( LLAMA_LOG_INFO("%s: pipeline parallelism enabled\n", __func__); } + if (cparams.moe_cache_size > 0) { + if (cparams.pipeline_parallel || model.n_devices() > 1) { + throw std::runtime_error("MoE cache does not support multiple devices"); + } + for (size_t i = 0; i < backend_ptrs.size(); ++i) { + const auto type = ggml_backend_dev_type(ggml_backend_get_device(backend_ptrs[i])); + if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) { + moe_cache = std::make_unique(model, backend_ptrs[i], backend_buft[i], cparams.moe_cache_size); + break; + } + } + if (!moe_cache) { + throw std::runtime_error("MoE cache requires a GPU backend"); + } + } + sched_reserve(); if (!cparams.flash_attn) { @@ -2605,6 +2623,7 @@ llm_graph_params llama_context::graph_params( /*.loras =*/ loras.get(), /*.mctx =*/ mctx, /*.cross =*/ &cross, + /*.moe_cache =*/ moe_cache.get(), /*.prec_policy =*/ &model.prec_policy, /*.samplers =*/ sampling.samplers, /*.n_outputs =*/ n_outputs, @@ -2645,7 +2664,14 @@ ggml_status llama_context::graph_compute( } bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data) { - auto & st = static_cast(user_data)->copy_experts; + auto * lctx = static_cast(user_data); + + // the slot maps of the MoE cache + if (lctx->moe_cache && lctx->moe_cache->copy(backend, src, dst, graph)) { + return true; + } + + auto & st = lctx->copy_experts; // the ids must be computed before the split starts, so only the first node of the split is considered if (ggml_graph_n_nodes(graph) == 0) { @@ -2692,10 +2718,28 @@ bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor last++; } + // the experts in the MoE cache are copied from device memory, the others are uploaded + int64_t next = first; + for (int64_t e = first; e <= last && lctx->moe_cache; ) { + const int64_t n = lctx->moe_cache->copy_experts(backend, src, dst, e, last); + if (n == 0) { + e++; + continue; + } + if (next < e) { + ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + next*expert_size, next*expert_size, (e - next)*expert_size); + } + e += n; + next = e; + } + // copy a bit extra to ensure there are no NaNs in the padding of the last expert, this is necessary for MMQ in the CUDA backend - const size_t offset = first*expert_size; + const size_t offset = next*expert_size; const size_t padding = last < n_expert - 1 ? std::min(expert_size, 512) : 0; - ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, (last - first + 1)*expert_size + padding); + const size_t size = (last + 1 - next)*expert_size + padding; + if (size > 0) { + ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, size); + } first = last + 1; } @@ -3562,6 +3606,11 @@ llama_memory_breakdown llama_context::memory_breakdown() const { ret[buft].context += size; } } + if (moe_cache) { + for (const auto & [buft, size] : moe_cache->memory_breakdown()) { + ret[buft].context += size; + } + } if (model.hparams.no_alloc) { for (size_t i = 0; i < backends.size(); ++i) { ggml_backend_t backend = backends[i].get(); @@ -3851,6 +3900,7 @@ llama_context_params llama_context_default_params() { /*.cb_eval_user_data =*/ nullptr, /*.type_k =*/ GGML_TYPE_F16, /*.type_v =*/ GGML_TYPE_F16, + /*.moe_cache_size =*/ 0, /*.abort_callback =*/ nullptr, /*.abort_callback_data =*/ nullptr, /*.embeddings =*/ false, diff --git a/src/llama-context.h b/src/llama-context.h index 28d386ad7e..69bc01c196 100644 --- a/src/llama-context.h +++ b/src/llama-context.h @@ -7,6 +7,7 @@ #include "llama-adapter.h" #include "llama-impl.h" #include "llama-memory.h" +#include "llama-moe-cache.h" #include "ggml-cpp.h" #include "ggml-opt.h" @@ -17,6 +18,7 @@ struct llama_model; class llama_batch_allocr; +class llama_moe_cache; class llama_io_read_i; class llama_io_write_i; @@ -271,7 +273,7 @@ private: llm_graph_cb graph_get_cb() const; - // ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID + // ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID and updates the MoE cache static bool sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data); // disable auto fused ops (Flash Attention, Gated Delta Net) whose op lands on a device @@ -299,6 +301,7 @@ private: llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably llama_memory_ptr memory; + llama_moe_cache_ptr moe_cache; // decode output (2-dimensional array: [n_outputs][n_vocab]) buffer_view logits = {nullptr, 0}; diff --git a/src/llama-cparams.h b/src/llama-cparams.h index 004fec5a6d..f4eb181a75 100644 --- a/src/llama-cparams.h +++ b/src/llama-cparams.h @@ -55,6 +55,8 @@ struct llama_cparams { bool pipeline_parallel; bool training; // set by llama_opt_init() + size_t moe_cache_size; + std::vector embeddings_layer_inp; // [n_layer()] extract input embeddings for layer enum llama_context_type ctx_type; diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp index 1112ad8854..e0b7a47a9b 100644 --- a/src/llama-graph.cpp +++ b/src/llama-graph.cpp @@ -2,6 +2,7 @@ #include "llama-impl.h" #include "llama-model.h" +#include "llama-moe-cache.h" #include "llama-batch.h" #include "llama-cparams.h" #include "llama-sampler.h" @@ -1523,6 +1524,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) : loras (params.loras), mctx (params.mctx), cross (params.cross), + moe_cache (params.moe_cache), prec_policy (params.prec_policy), samplers (params.samplers), cb_func (params.cb), @@ -1589,8 +1591,12 @@ ggml_tensor * llm_graph_context::build_lora_mm_id( ggml_tensor * w, // ggml_tensor * as ggml_tensor * cur, // ggml_tensor * b ggml_tensor * ids, - ggml_tensor * w_s) const { - ggml_tensor * res = ggml_mul_mat_id(ctx0, w, cur, ids); + ggml_tensor * w_s, + ggml_tensor * slots) const { + // the experts in the MoE cache are selected by their slots + ggml_tensor * res = slots == nullptr ? + ggml_mul_mat_id(ctx0, w, cur, ids) : + ggml_mul_mat_id(ctx0, moe_cache->get_experts(w), cur, slots); if (prec_policy) { prec_policy->apply(res); @@ -2205,6 +2211,9 @@ ggml_tensor * llm_graph_context::build_moe_ffn( //call early so that topk-moe can be used ggml_build_forward_expand(gf, weights); + // the experts of host-resident layers may be read from the MoE cache + ggml_tensor * slots = build_moe_cache_slots(selected_experts, up_exps, gate_exps, down_exps, gate_up_exps, il); + cur = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens); if (weight_before_ffn) { @@ -2219,7 +2228,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn( if (gate_up_exps) { // merged gate_up path: one mul_mat_id, then split into gate and up views - ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s); // [n_ff*2, n_expert_used, n_tokens] + ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff*2, n_expert_used, n_tokens] cb(gate_up, "ffn_moe_gate_up", il); if (up_exps_s) { @@ -2238,7 +2247,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn( cb(up, "ffn_moe_up", il); } else { // separate gate and up path - up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s); // [n_ff, n_expert_used, n_tokens] + up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff, n_expert_used, n_tokens] cb(up, "ffn_moe_up", il); if (up_exps_s) { @@ -2251,7 +2260,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn( } if (gate_exps) { - cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s); // [n_ff, n_expert_used, n_tokens] + cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s, slots); // [n_ff, n_expert_used, n_tokens] cb(cur, "ffn_moe_gate", il); } else { cur = up; @@ -2352,7 +2361,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn( GGML_ABORT("fatal error"); } - experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s); // [n_embd, n_expert_used, n_tokens] + experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s, slots); // [n_embd, n_expert_used, n_tokens] if (arch == LLM_ARCH_MISTRAL4) { // src1 can exceed F16 range ggml_prec_set_src(experts, GGML_PREC_F32, 1); @@ -2409,6 +2418,45 @@ ggml_tensor * llm_graph_context::build_moe_ffn( return moe_out; } +ggml_tensor * llm_graph_context::build_moe_cache_slots( + ggml_tensor * selected_experts, + ggml_tensor * up_exps, + ggml_tensor * gate_exps, + ggml_tensor * down_exps, + ggml_tensor * gate_up_exps, + int il) const { + if (moe_cache == nullptr) { + return nullptr; + } + + ggml_tensor * slot_map = moe_cache->get_slot_map(il, selected_experts->ne[1], selected_experts->ne[0]); + if (slot_map == nullptr) { + return nullptr; + } + for (ggml_tensor * w : { up_exps, gate_exps, down_exps, gate_up_exps }) { + if (w != nullptr && moe_cache->get_experts(w) == nullptr) { + return nullptr; + } + } + + ggml_tensor * ids = selected_experts; + if (!ggml_is_contiguous(ids)) { + ids = ggml_cont(ctx0, ids); + } + ids = ggml_reshape_1d(ctx0, ids, ggml_nelements(ids)); + + // the slot map is a host weight, so the scheduler starts a new split here and copies it with the copy callback + // the callback reads the selected experts, uploads the missing ones and updates the slot map + ggml_tensor * slots = ggml_get_rows(ctx0, slot_map, ids); // [1, n_expert_used*n_tokens] + if (!ggml_backend_supports_op(moe_cache->backend(), slots)) { + return nullptr; + } + ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend()); + cb(slots, "ffn_moe_slots", il); + + return ggml_reshape_2d(ctx0, slots, selected_experts->ne[0], selected_experts->ne[1]); // [n_expert_used, n_tokens] +} + // input embeddings with optional lora ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float tok_scale) const { const int64_t n_embd_inp = hparams.n_embd_inp(); diff --git a/src/llama-graph.h b/src/llama-graph.h index 838544576f..79e8409ac8 100644 --- a/src/llama-graph.h +++ b/src/llama-graph.h @@ -21,6 +21,8 @@ struct llama_cparams; struct llama_layer; struct llama_prec_policy; +class llama_moe_cache; + struct llama_memory_context_i; class llama_kv_cache_context; @@ -793,6 +795,7 @@ struct llm_graph_params { const llama_adapter_loras * loras; const llama_memory_context_i * mctx; const llama_cross * cross; + const llama_moe_cache * moe_cache; const llama_prec_policy * prec_policy = nullptr; @@ -1036,6 +1039,7 @@ struct llm_graph_context { const llama_adapter_loras * loras; const llama_memory_context_i * mctx; const llama_cross * cross; + const llama_moe_cache * moe_cache; const llama_prec_policy * prec_policy; @@ -1078,11 +1082,13 @@ struct llm_graph_context { ggml_tensor * w_s = nullptr) const; // do mat_mul_id, while optionally apply lora and per-expert scale + // if slots is set, the experts are read from the MoE cache at these slots (see build_moe_cache_slots) ggml_tensor * build_lora_mm_id( ggml_tensor * w, // ggml_tensor * as ggml_tensor * cur, // ggml_tensor * b ggml_tensor * ids, - ggml_tensor * w_s = nullptr) const; + ggml_tensor * w_s = nullptr, + ggml_tensor * slots = nullptr) const; ggml_tensor * build_norm( ggml_tensor * cur, @@ -1179,6 +1185,15 @@ struct llm_graph_context { ggml_tensor * down_exps_s = nullptr, ggml_tensor * selected_experts_in = nullptr) const; + // the slots of the selected experts in the MoE cache, nullptr if the experts of the layer are not read from the cache + ggml_tensor * build_moe_cache_slots( + ggml_tensor * selected_experts, + ggml_tensor * up_exps, + ggml_tensor * gate_exps, + ggml_tensor * down_exps, + ggml_tensor * gate_up_exps, + int il) const; + // // inputs // diff --git a/src/llama-moe-cache.cpp b/src/llama-moe-cache.cpp new file mode 100644 index 0000000000..7292a1e570 --- /dev/null +++ b/src/llama-moe-cache.cpp @@ -0,0 +1,550 @@ +#include "llama-moe-cache.h" + +#include "llama-impl.h" +#include "llama-model.h" + +#include "ggml-cpp.h" + +#include +#include +#include +#include + +namespace { + +// LRU of the experts of a group of layers, the slot of each expert is kept in the slot map of its layer +struct moe_cache_lru { + int32_t n_expert = 0; + int32_t n_slots = 0; + + std::vector slot_map; // [n_layer] data of the slot maps, -1 if the expert is not cached + std::vector key_of; // [n_slots] il*n_expert + expert, -1 if empty + + // doubly linked list of the slots, head is the least recently used + std::vector prev; + std::vector next; + int32_t head = -1; + int32_t tail = -1; + + std::vector seen; // [n_expert] + uint32_t seen_gen = 0; + std::vector uniq; + + void init(int32_t n_layer, int32_t n_expert, int32_t n_slots) { + this->n_expert = n_expert; + this->n_slots = n_slots; + slot_map.assign(n_layer, nullptr); + key_of.assign(n_slots, -1); + prev.resize(n_slots); + next.resize(n_slots); + for (int32_t s = 0; s < n_slots; ++s) { + prev[s] = s - 1; + next[s] = s + 1 < n_slots ? s + 1 : -1; + } + head = 0; + tail = n_slots - 1; + seen.assign(n_expert, 0); + } + + // move slot s to the tail (most recently used) + void touch(int32_t s) { + if (s == tail) { + return; + } + if (prev[s] >= 0) { + next[prev[s]] = next[s]; + } else { + head = next[s]; + } + prev[next[s]] = prev[s]; + + prev[s] = tail; + next[s] = -1; + next[tail] = s; + tail = s; + } + + struct fill { + int32_t expert; + int32_t slot; + }; + + // give a slot to each expert selected by ids in layer il, the misses evict the least recently used experts + // returns false if the ids select more distinct experts than there are slots + bool plan(int32_t il, const int32_t * ids, size_t n_ids, std::vector & fills, size_t & n_hit) { + fills.clear(); + n_hit = 0; + + if (++seen_gen == 0) { + std::fill(seen.begin(), seen.end(), 0); + seen_gen = 1; + } + uniq.clear(); + for (size_t i = 0; i < n_ids; ++i) { + GGML_ASSERT(ids[i] >= 0 && ids[i] < n_expert); + if (seen[ids[i]] != seen_gen) { + seen[ids[i]] = seen_gen; + uniq.push_back(ids[i]); + } + } + if (uniq.size() > (size_t) n_slots) { + return false; + } + + int32_t * slots = slot_map[il]; + + // hits go to the tail first, so the head can be evicted below + for (int32_t e : uniq) { + if (slots[e] >= 0) { + touch(slots[e]); + n_hit++; + } + } + // sorted misses usually get consecutive slots, so the uploads can be merged + std::sort(uniq.begin(), uniq.end()); + for (int32_t e : uniq) { + if (slots[e] >= 0) { + continue; + } + const int32_t s = head; + if (key_of[s] >= 0) { + slot_map[key_of[s] / n_expert][key_of[s] % n_expert] = -1; + } + key_of[s] = il*n_expert + e; + slots[e] = s; + touch(s); + fills.push_back({ e, s }); + } + return true; + } +}; + +// gate, up, down or gate_up, down +static std::vector llama_moe_cache_layer_experts(const llama_layer & layer) { + std::vector res; + for (ggml_tensor * t : { layer.ffn_gate_up_exps, layer.ffn_gate_exps, layer.ffn_up_exps, layer.ffn_down_exps }) { + if (t != nullptr) { + res.push_back(t); + } + } + return res; +} + +static bool llama_moe_cache_same_layout(const std::vector & a, const std::vector & b) { + if (a.size() != b.size()) { + return false; + } + for (size_t i = 0; i < a.size(); ++i) { + if (a[i]->type != b[i]->type || !ggml_are_same_shape(a[i], b[i]) || a[i]->nb[2] != b[i]->nb[2]) { + return false; + } + } + return true; +} + +static bool llama_moe_cache_is_host_weight(const ggml_tensor * t) { + return t->buffer != nullptr && + ggml_backend_buffer_get_usage(t->buffer) == GGML_BACKEND_BUFFER_USAGE_WEIGHTS && + ggml_backend_buffer_is_host(t->buffer); +} + +} + +struct llama_moe_cache::impl { + // layers with the same expert tensor layout share the banks and the LRU of a group + struct group { + std::vector ref; // expert tensors of the first layer + std::vector layers; + std::vector banks; // device storage of all slots, one per expert tensor + size_t host_bytes = 0; + int32_t n_slots = 0; + moe_cache_lru lru; + }; + + struct layer { + int32_t ig = -1; // -1 if the layer is not cached + ggml_tensor * slot_map = nullptr; // I32 [1, n_expert] in host memory + std::vector experts; // host expert tensors, in the order of the banks + }; + + struct binding { + int32_t il; + int32_t ip; // index of the bank + ggml_tensor * cached; // view of the bank used in place of the host experts + }; + + struct stats { + size_t hits = 0; + size_t misses = 0; + size_t bytes = 0; + }; + + static constexpr int64_t max_batch = 32; + + ggml_backend_t backend; + int32_t n_expert_used; + + stats stats_small; // up to 8 tokens per ubatch + stats stats_large; + stats stats_copy; // experts copied from the cache for large batches + + std::vector groups; + std::vector layers; + std::unordered_map bindings; // host experts -> cached experts + std::unordered_map layer_of; // slot map -> layer + + std::vector ids; + std::vector fills; + + // banks and their views on the device + ggml_context_ptr ctx; + ggml_backend_buffer_ptr buf; + size_t buf_size = 0; + + // slot maps in host memory + ggml_context_ptr ctx_host; + ggml_backend_buffer_ptr buf_host; + size_t buf_host_size = 0; + + // views used by copy_experts + ggml_context_ptr ctx_views; + + impl(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) : + backend(backend), n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) { + ggml_backend_dev_t dev = ggml_backend_get_device(backend); + const auto dev_type = ggml_backend_dev_type(dev); + if (dev_type != GGML_BACKEND_DEVICE_TYPE_GPU && dev_type != GGML_BACKEND_DEVICE_TYPE_IGPU) { + throw std::runtime_error("MoE cache requires a GPU backend"); + } + if (model.split_mode() == LLAMA_SPLIT_MODE_TENSOR) { + throw std::runtime_error("MoE cache does not support tensor parallelism"); + } + if (model.hparams.n_expert == 0 || n_expert_used == 0) { + throw std::runtime_error("MoE cache requires a MoE model"); + } + + // only cache layers that keep all of their experts in host memory + size_t host_bytes = 0; + for (size_t il = 0; il < model.layers.size(); ++il) { + auto experts = llama_moe_cache_layer_experts(model.layers[il]); + if (experts.empty() || model.dev_layer(il) != dev || + !std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) { + continue; + } + auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return llama_moe_cache_same_layout(g.ref, experts); }); + if (it == groups.end()) { + groups.emplace_back(); + it = groups.end() - 1; + it->ref = experts; + } + it->layers.push_back(il); + for (const ggml_tensor * t : experts) { + it->host_bytes += ggml_nbytes(t); + host_bytes += ggml_nbytes(t); + } + } + if (groups.empty()) { + LLAMA_LOG_WARN("%s: no layer has all of its experts in host memory, MoE cache is disabled\n", __func__); + return; + } + + // one extra slot at the end, CUDA MMQ can read past the last expert + const size_t alignment = ggml_backend_buft_get_alignment(buft); + auto alloc_size = [&](const group & g, int32_t n_slots) { + size_t res = 0; + for (const ggml_tensor * t : g.ref) { + res += GGML_PAD(t->nb[2]*(n_slots + 1), alignment); + } + return res; + }; + + // split the budget by the size of the experts, so each group caches the same fraction of its experts + size_t n_tensors = 0; + size_t n_tensors_host = 0; + for (group & g : groups) { + const int32_t n_expert = g.ref[0]->ne[2]; + const size_t budget = (size_t) ((double) size*g.host_bytes/host_bytes); + const int32_t max_slots = g.layers.size()*n_expert; + while (g.n_slots < max_slots && alloc_size(g, g.n_slots + 1) <= budget) { + g.n_slots++; + } + if (g.n_slots < n_expert_used) { + LLAMA_LOG_WARN("%s: MoE cache budget is too small for %zu layers, they are not cached\n", __func__, g.layers.size()); + g.n_slots = 0; + continue; + } + g.lru.init(model.layers.size(), n_expert, g.n_slots); + n_tensors += g.ref.size()*(1 + g.layers.size()); + n_tensors_host += g.layers.size(); + } + if (n_tensors == 0) { + throw std::runtime_error("MoE cache is too small to hold the experts of one token"); + } + + auto init_ctx = [](size_t n_tensors) { + ggml_init_params params = { + /*.mem_size =*/ n_tensors*ggml_tensor_overhead(), + /*.mem_buffer =*/ nullptr, + /*.no_alloc =*/ true, + }; + ggml_context_ptr res(ggml_init(params)); + if (!res) { + throw std::runtime_error("failed to create the MoE cache context"); + } + return res; + }; + ctx = init_ctx(n_tensors); + ctx_host = init_ctx(n_tensors_host); + ctx_views = init_ctx(2); + + ggml_backend_buffer_type_t buft_host = ggml_backend_cpu_buffer_type(); + const size_t alignment_host = ggml_backend_buft_get_alignment(buft_host); + + for (size_t ig = 0; ig < groups.size(); ++ig) { + group & g = groups[ig]; + if (g.n_slots == 0) { + continue; + } + for (const ggml_tensor * t : g.ref) { + ggml_tensor * bank = ggml_new_tensor_3d(ctx.get(), t->type, t->ne[0], t->ne[1], g.n_slots + 1); + GGML_ASSERT(bank->nb[2] == t->nb[2]); + ggml_format_name(bank, "moe_cache.%zu.%s", ig, t->name); + g.banks.push_back(bank); + } + for (int32_t il : g.layers) { + layer & l = layers[il]; + l.ig = (int32_t) ig; + l.experts = llama_moe_cache_layer_experts(model.layers[il]); + for (size_t ip = 0; ip < l.experts.size(); ++ip) { + ggml_tensor * bank = g.banks[ip]; + ggml_tensor * cached = ggml_view_3d(ctx.get(), bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0); + ggml_format_name(cached, "moe_cache.%s", l.experts[ip]->name); + bindings[l.experts[ip]] = { il, (int32_t) ip, cached }; + } + l.slot_map = ggml_new_tensor_2d(ctx_host.get(), GGML_TYPE_I32, 1, g.ref[0]->ne[2]); + ggml_format_name(l.slot_map, "moe_cache.slot_map-%d", il); + layer_of[l.slot_map] = il; + buf_host_size += GGML_PAD(ggml_nbytes(l.slot_map), alignment_host); + } + buf_size += alloc_size(g, g.n_slots); + } + + if (model.hparams.no_alloc) { + // only used to measure the memory use, see llama_context::memory_breakdown + buf.reset(ggml_backend_buft_alloc_buffer(buft, 0)); + buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0)); + for (ggml_tensor * t = ggml_get_first_tensor(ctx.get()); t != nullptr; t = ggml_get_next_tensor(ctx.get(), t)) { + t->buffer = buf.get(); + } + for (ggml_tensor * t = ggml_get_first_tensor(ctx_host.get()); t != nullptr; t = ggml_get_next_tensor(ctx_host.get(), t)) { + t->buffer = buf_host.get(); + } + } else { + buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx.get(), buft)); + buf_host.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx_host.get(), buft_host)); + if (!buf || !buf_host) { + throw std::runtime_error("failed to allocate the MoE cache buffers"); + } + ggml_backend_buffer_clear(buf.get(), 0); + ggml_backend_buffer_clear(buf_host.get(), 0xff); // all slots are -1 + buf_size = ggml_backend_buffer_get_size(buf.get()); + buf_host_size = ggml_backend_buffer_get_size(buf_host.get()); + + for (group & g : groups) { + for (int32_t il : g.layers) { + if (layers[il].slot_map != nullptr) { + g.lru.slot_map[il] = (int32_t *) layers[il].slot_map->data; + } + } + } + } + + // as weights, the ops that read the banks run on the device and the slot maps are copied with the copy callback + ggml_backend_buffer_set_usage(buf.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS); + ggml_backend_buffer_set_usage(buf_host.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS); + + LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__, + ggml_backend_buft_name(buft), buf_size/1024.0/1024.0, host_bytes/1024.0/1024.0); + for (const group & g : groups) { + LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__, + g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2])); + } + } + + ~impl() { + log_stats(); + } + + ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const { + if (il < 0 || il >= (int32_t) layers.size() || layers[il].ig < 0) { + return nullptr; + } + const layer & l = layers[il]; + + // large batches use most experts of a layer, so they gain little from the cache and would evict the experts used in generation + if (n_tokens == 0 || n_tokens > max_batch || std::min(n_tokens*n_expert_used, l.slot_map->ne[1]) > groups[l.ig].n_slots) { + return nullptr; + } + return l.slot_map; + } + + ggml_tensor * get_experts(const ggml_tensor * w) const { + const auto it = bindings.find(w); + return it != bindings.end() ? it->second.cached : nullptr; + } + + int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) { + const auto it = bindings.find(w); + if (it == bindings.end() || backend != this->backend) { + return 0; + } + const binding & b = it->second; + const group & g = groups[layers[b.il].ig]; + + // large batches only read the cache, so the experts used in generation stay in it + const int32_t * slots = g.lru.slot_map[b.il]; + if (slots == nullptr || slots[e] < 0) { + return 0; + } + int64_t n = 1; + while (e + n <= last && slots[e + n] == slots[e] + n) { + n++; + } + + ggml_tensor * bank = g.banks[b.ip]; + ggml_reset(ctx_views.get()); + ggml_tensor * src_view = ggml_view_3d(ctx_views.get(), bank, bank->ne[0], bank->ne[1], n, bank->nb[1], bank->nb[2], slots[e]*bank->nb[2]); + ggml_tensor * dst_view = ggml_view_3d(ctx_views.get(), dst, dst->ne[0], dst->ne[1], n, dst->nb[1], dst->nb[2], e*dst->nb[2]); + ggml_backend_view_init(src_view); + ggml_backend_view_init(dst_view); + ggml_backend_tensor_copy_async(backend, backend, src_view, dst_view); + + stats_copy.hits += n; + stats_copy.bytes += ggml_nbytes(src_view); + + return n; + } + + bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) { + const auto it = layer_of.find(src); + if (it == layer_of.end()) { + return false; + } + const int32_t il = it->second; + const layer & l = layers[il]; + group & g = groups[l.ig]; + + GGML_ASSERT(backend == this->backend); + + // the get_rows that looks up the slots of the selected experts + const int n_nodes = ggml_graph_n_nodes(graph); + const ggml_tensor * lookup = nullptr; + for (int i = 0; i < n_nodes && lookup == nullptr; ++i) { + const ggml_tensor * node = ggml_graph_node(graph, i); + if (node->op == GGML_OP_GET_ROWS && node->src[0] == dst) { + lookup = node; + } + } + GGML_ASSERT(lookup != nullptr); + + // the selected experts must be computed in an earlier split + // the scheduler starts a new split at the lookup because it reads a host weight, but only if the split already has inputs + const ggml_tensor * sel = lookup->src[1]; + for (int i = 0; i < n_nodes; ++i) { + const ggml_tensor * node = ggml_graph_node(graph, i); + if (node == sel || node == sel->view_src) { + GGML_ABORT("the experts of layer %d are selected in the same split as their MoE cache lookup", il); + } + } + GGML_ASSERT(ggml_is_contiguous(sel)); + + ids.resize(ggml_nelements(sel)); + ggml_backend_tensor_get_async(backend, sel, ids.data(), 0, ggml_nbytes(sel)); + ggml_backend_synchronize(backend); + + size_t n_hit = 0; + if (!g.lru.plan(il, ids.data(), ids.size(), fills, n_hit)) { + GGML_ABORT("the MoE cache is too small for the experts selected in layer %d", il); + } + + // upload the missing experts, consecutive experts going to consecutive slots are uploaded together + size_t bytes = 0; + for (size_t ip = 0; ip < l.experts.size(); ++ip) { + const ggml_tensor * w = l.experts[ip]; + ggml_tensor * bank = g.banks[ip]; + const size_t expert_size = w->nb[2]; + for (size_t i = 0; i < fills.size();) { + size_t n = 1; + while (i + n < fills.size() && fills[i + n].expert == fills[i].expert + (int32_t) n && fills[i + n].slot == fills[i].slot + (int32_t) n) { + n++; + } + ggml_backend_tensor_set_async(backend, bank, (const uint8_t *) w->data + fills[i].expert*expert_size, fills[i].slot*expert_size, n*expert_size); + bytes += n*expert_size; + i += n; + } + } + + stats & st = ids.size() <= (size_t) 8*n_expert_used ? stats_small : stats_large; + st.hits += n_hit; + st.misses += fills.size(); + st.bytes += bytes; + + // the next copy synchronizes the backend before it changes the slot map again + ggml_backend_tensor_set_async(backend, dst, src->data, 0, ggml_nbytes(src)); + + return true; + } + + void log_stats() const { + auto log = [](const char * name, const stats & st) { + const size_t n = st.hits + st.misses; + if (n == 0) { + return; + } + LLAMA_LOG_INFO("llama_moe_cache: %s: hits = %zu, misses = %zu, hit rate = %.2f%%, uploaded = %.2f MiB\n", + name, st.hits, st.misses, 100.0*st.hits/n, st.bytes/1024.0/1024.0); + }; + log("ubatch <= 8", stats_small); + log("ubatch > 8", stats_large); + if (stats_copy.hits > 0) { + LLAMA_LOG_INFO("llama_moe_cache: large batches: %zu experts copied from the cache, %.2f MiB\n", stats_copy.hits, stats_copy.bytes/1024.0/1024.0); + } + } +}; + +llama_moe_cache::llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) : + pimpl(new impl(model, backend, buft, size)) { +} + +llama_moe_cache::~llama_moe_cache() = default; + +ggml_backend_t llama_moe_cache::backend() const { + return pimpl->backend; +} + +ggml_tensor * llama_moe_cache::get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const { + return pimpl->get_slot_map(il, n_tokens, n_expert_used); +} + +ggml_tensor * llama_moe_cache::get_experts(const ggml_tensor * w) const { + return pimpl->get_experts(w); +} + +bool llama_moe_cache::copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) { + return pimpl->copy(backend, src, dst, graph); +} + +int64_t llama_moe_cache::copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) { + return pimpl->copy_experts(backend, w, dst, e, last); +} + +std::map llama_moe_cache::memory_breakdown() const { + std::map res; + if (pimpl->buf) { + res[ggml_backend_buffer_get_type(pimpl->buf.get())] += pimpl->buf_size; + } + if (pimpl->buf_host) { + res[ggml_backend_buffer_get_type(pimpl->buf_host.get())] += pimpl->buf_host_size; + } + return res; +} diff --git a/src/llama-moe-cache.h b/src/llama-moe-cache.h new file mode 100644 index 0000000000..4872370d46 --- /dev/null +++ b/src/llama-moe-cache.h @@ -0,0 +1,39 @@ +#pragma once + +#include "ggml-backend.h" + +#include +#include + +struct llama_model; + +// keeps the most recently used experts of host-resident MoE layers in a device buffer +// each layer has a slot map in host memory: when the scheduler copies it to the device, the copy callback uploads the missing experts +class llama_moe_cache { +public: + llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size); + ~llama_moe_cache(); + + ggml_backend_t backend() const; + + // the slot map of layer il, if its experts can be read from the cache for n_tokens tokens, nullptr otherwise + ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const; + + // the experts of w in the cache, nullptr if w is not cached + ggml_tensor * get_experts(const ggml_tensor * w) const; + + // ggml_backend_sched copy callback, returns false if src is not a slot map + bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph); + + // for large batches: copy the experts of w that are in the cache, starting at expert e and up to expert last, to the copy dst of w + // returns the number of experts copied, 0 if expert e is not in the cache + int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last); + + std::map memory_breakdown() const; + +private: + struct impl; + std::unique_ptr pimpl; +}; + +using llama_moe_cache_ptr = std::unique_ptr; diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index 54122629ae..b54b3cc18c 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -495,7 +495,7 @@ static std::pair get_model_and_ctx( struct gguf_context * gguf_ctx, FILE * file, const size_t seed, const float stdev, const std::vector & devs, const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false, - const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr) { + const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr, const size_t moe_cache_size = 0) { GGML_ASSERT((gguf_ctx == nullptr) != (file == nullptr)); llama_model_params model_params = llama_model_default_params(); model_params.progress_callback = silent_model_load_progress; @@ -512,6 +512,11 @@ static std::pair get_model_and_ctx( if (!encode) { ctx_params.n_ubatch = 64; } + if (moe_cache_size > 0) { + // the MoE cache is only used for small ubatches + ctx_params.moe_cache_size = moe_cache_size; + ctx_params.n_ubatch = 2; + } tensor_data_params tensor_params = { seed, stdev }; llama_model_ptr model(gguf_ctx != nullptr ? @@ -865,9 +870,10 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con std::string label; llama_split_mode split_mode; bool host_experts; // keep the experts in host memory, see host_experts_test + size_t moe_cache_size; - device_config(std::vector devs, std::string name, llama_split_mode split_mode, bool host_experts = false) - : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts) {} + device_config(std::vector devs, std::string name, llama_split_mode split_mode, bool host_experts = false, size_t moe_cache_size = 0) + : devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts), moe_cache_size(moe_cache_size) {} }; const llama_model_tensor_buft_override host_experts_overrides[] = { @@ -905,6 +911,16 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con dev_configs.emplace_back(devices_meta, "Host experts", LLAMA_SPLIT_MODE_LAYER, true); max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length()); } + + // the ops that use the host experts run on a GPU and read the experts from a cache + // the cache has only a few slots (4 for 288 KiB experts), so the experts are evicted and uploaded again + if (!devices_meta.empty()) { + const enum ggml_backend_dev_type type = ggml_backend_dev_type(devices_meta[0]); + if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) { + dev_configs.emplace_back(std::vector{devices_meta[0]}, "MoE cache", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024); + max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length()); + } + } } size_t max_arch_name_length = 0; @@ -987,7 +1003,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con } if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) { test_executed = true; - model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides); + model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size); logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode); const double nmse_val = nmse(logits_cpu, logits_dev); snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val); @@ -1053,7 +1069,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con ms.save(file); rewind(file); - auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides); + auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size); const std::vector logits_roundtrip = get_logits( model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode); status_roundtrip = "\033[1;32mOK\033[0m";