mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-10 23:10:48 +02:00
llama : add a GPU cache for MoE experts kept in host memory (#29887)
* llama : add a GPU cache for MoE experts kept in host memory Assisted-by: Claude * use llama_moe_cache_ptr
This commit is contained in:
@@ -2776,6 +2776,16 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
llm_add_n_cpu_ffn_overrides(value, LLM_FFN_EXPS_REGEX, params.tensor_buft_overrides);
|
||||
}
|
||||
).set_env("LLAMA_ARG_N_CPU_MOE"));
|
||||
add_opt(common_arg(
|
||||
{"--moe-cache-mib"}, "N",
|
||||
"GPU cache size in MiB for the MoE experts kept in the CPU (default: 0, disabled)",
|
||||
[](common_params & params, int value) {
|
||||
if (value < 0) {
|
||||
throw std::invalid_argument("invalid value");
|
||||
}
|
||||
params.moe_cache_size = (size_t) value*1024*1024;
|
||||
}
|
||||
).set_env("LLAMA_ARG_MOE_CACHE_MIB"));
|
||||
add_opt(common_arg(
|
||||
{"-ncffn", "--n-cpu-ffn"}, "N",
|
||||
"keep the dense FFN weights of the first N layers in the CPU\n"
|
||||
|
||||
@@ -1722,6 +1722,8 @@ struct llama_context_params common_context_params_to_llama(const common_params &
|
||||
cparams.type_k = params.cache_type_k;
|
||||
cparams.type_v = params.cache_type_v;
|
||||
|
||||
cparams.moe_cache_size = params.moe_cache_size;
|
||||
|
||||
return cparams;
|
||||
}
|
||||
|
||||
|
||||
@@ -593,6 +593,8 @@ struct common_params {
|
||||
ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K
|
||||
ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V
|
||||
|
||||
size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU
|
||||
|
||||
common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;
|
||||
|
||||
// multimodal models (see tools/mtmd)
|
||||
|
||||
@@ -2561,6 +2561,9 @@ common_params common_base_params_to_speculative(const common_params & params) {
|
||||
result.n_outputs_max = params.n_parallel;
|
||||
result.n_outputs_max_per_seq = 1;
|
||||
|
||||
// the MoE cache is only used by the target context
|
||||
result.moe_cache_size = 0;
|
||||
|
||||
// dflash/dspark decode the whole noise block in a single pass and sample every block position on the backend
|
||||
// TODO: refactor such properties to be announced by the speculative types
|
||||
// something like `struct common_speculative_type_props common_speculative_type_get_props(...);`
|
||||
|
||||
@@ -396,6 +396,8 @@ extern "C" {
|
||||
enum ggml_type type_k; // data type for K cache [EXPERIMENTAL]
|
||||
enum ggml_type type_v; // data type for V cache [EXPERIMENTAL]
|
||||
|
||||
size_t moe_cache_size; // device cache in bytes for the experts kept in host memory, 0 = disabled [EXPERIMENTAL]
|
||||
|
||||
// Abort callback
|
||||
// if it returns true, execution of llama_decode() will be aborted
|
||||
// currently works only with CPU execution
|
||||
|
||||
@@ -28,6 +28,7 @@ set(LLAMA_CORE_SOURCES
|
||||
llama-kv-cache-msa.cpp
|
||||
llama-kv-cache-dsv4.cpp
|
||||
llama-memory.cpp
|
||||
llama-moe-cache.cpp
|
||||
llama-memory-hybrid.cpp
|
||||
llama-memory-hybrid-iswa.cpp
|
||||
llama-memory-hybrid-idx.cpp
|
||||
|
||||
+53
-3
@@ -9,6 +9,7 @@
|
||||
#include "llama-memory.h"
|
||||
#include "llama-mmap.h"
|
||||
#include "llama-model.h"
|
||||
#include "llama-moe-cache.h"
|
||||
#include "llama-ext.h"
|
||||
#include "llama-sampler.h"
|
||||
#include "llama.h"
|
||||
@@ -274,6 +275,7 @@ llama_context::llama_context(
|
||||
|
||||
cparams.op_offload = params.op_offload;
|
||||
cparams.kv_unified = params.kv_unified;
|
||||
cparams.moe_cache_size = params.moe_cache_size;
|
||||
|
||||
// initialized later
|
||||
cparams.pipeline_parallel = false;
|
||||
@@ -462,6 +464,22 @@ llama_context::llama_context(
|
||||
LLAMA_LOG_INFO("%s: pipeline parallelism enabled\n", __func__);
|
||||
}
|
||||
|
||||
if (cparams.moe_cache_size > 0) {
|
||||
if (cparams.pipeline_parallel || model.n_devices() > 1) {
|
||||
throw std::runtime_error("MoE cache does not support multiple devices");
|
||||
}
|
||||
for (size_t i = 0; i < backend_ptrs.size(); ++i) {
|
||||
const auto type = ggml_backend_dev_type(ggml_backend_get_device(backend_ptrs[i]));
|
||||
if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
|
||||
moe_cache = std::make_unique<llama_moe_cache>(model, backend_ptrs[i], backend_buft[i], cparams.moe_cache_size);
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!moe_cache) {
|
||||
throw std::runtime_error("MoE cache requires a GPU backend");
|
||||
}
|
||||
}
|
||||
|
||||
sched_reserve();
|
||||
|
||||
if (!cparams.flash_attn) {
|
||||
@@ -2605,6 +2623,7 @@ llm_graph_params llama_context::graph_params(
|
||||
/*.loras =*/ loras.get(),
|
||||
/*.mctx =*/ mctx,
|
||||
/*.cross =*/ &cross,
|
||||
/*.moe_cache =*/ moe_cache.get(),
|
||||
/*.prec_policy =*/ &model.prec_policy,
|
||||
/*.samplers =*/ sampling.samplers,
|
||||
/*.n_outputs =*/ n_outputs,
|
||||
@@ -2645,7 +2664,14 @@ ggml_status llama_context::graph_compute(
|
||||
}
|
||||
|
||||
bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data) {
|
||||
auto & st = static_cast<llama_context *>(user_data)->copy_experts;
|
||||
auto * lctx = static_cast<llama_context *>(user_data);
|
||||
|
||||
// the slot maps of the MoE cache
|
||||
if (lctx->moe_cache && lctx->moe_cache->copy(backend, src, dst, graph)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
auto & st = lctx->copy_experts;
|
||||
|
||||
// the ids must be computed before the split starts, so only the first node of the split is considered
|
||||
if (ggml_graph_n_nodes(graph) == 0) {
|
||||
@@ -2692,10 +2718,28 @@ bool llama_context::sched_copy_experts(ggml_backend_t backend, const ggml_tensor
|
||||
last++;
|
||||
}
|
||||
|
||||
// the experts in the MoE cache are copied from device memory, the others are uploaded
|
||||
int64_t next = first;
|
||||
for (int64_t e = first; e <= last && lctx->moe_cache; ) {
|
||||
const int64_t n = lctx->moe_cache->copy_experts(backend, src, dst, e, last);
|
||||
if (n == 0) {
|
||||
e++;
|
||||
continue;
|
||||
}
|
||||
if (next < e) {
|
||||
ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + next*expert_size, next*expert_size, (e - next)*expert_size);
|
||||
}
|
||||
e += n;
|
||||
next = e;
|
||||
}
|
||||
|
||||
// copy a bit extra to ensure there are no NaNs in the padding of the last expert, this is necessary for MMQ in the CUDA backend
|
||||
const size_t offset = first*expert_size;
|
||||
const size_t offset = next*expert_size;
|
||||
const size_t padding = last < n_expert - 1 ? std::min<size_t>(expert_size, 512) : 0;
|
||||
ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, (last - first + 1)*expert_size + padding);
|
||||
const size_t size = (last + 1 - next)*expert_size + padding;
|
||||
if (size > 0) {
|
||||
ggml_backend_tensor_set_async(backend, dst, (const uint8_t *) src->data + offset, offset, size);
|
||||
}
|
||||
|
||||
first = last + 1;
|
||||
}
|
||||
@@ -3562,6 +3606,11 @@ llama_memory_breakdown llama_context::memory_breakdown() const {
|
||||
ret[buft].context += size;
|
||||
}
|
||||
}
|
||||
if (moe_cache) {
|
||||
for (const auto & [buft, size] : moe_cache->memory_breakdown()) {
|
||||
ret[buft].context += size;
|
||||
}
|
||||
}
|
||||
if (model.hparams.no_alloc) {
|
||||
for (size_t i = 0; i < backends.size(); ++i) {
|
||||
ggml_backend_t backend = backends[i].get();
|
||||
@@ -3851,6 +3900,7 @@ llama_context_params llama_context_default_params() {
|
||||
/*.cb_eval_user_data =*/ nullptr,
|
||||
/*.type_k =*/ GGML_TYPE_F16,
|
||||
/*.type_v =*/ GGML_TYPE_F16,
|
||||
/*.moe_cache_size =*/ 0,
|
||||
/*.abort_callback =*/ nullptr,
|
||||
/*.abort_callback_data =*/ nullptr,
|
||||
/*.embeddings =*/ false,
|
||||
|
||||
+4
-1
@@ -7,6 +7,7 @@
|
||||
#include "llama-adapter.h"
|
||||
#include "llama-impl.h"
|
||||
#include "llama-memory.h"
|
||||
#include "llama-moe-cache.h"
|
||||
|
||||
#include "ggml-cpp.h"
|
||||
#include "ggml-opt.h"
|
||||
@@ -17,6 +18,7 @@
|
||||
|
||||
struct llama_model;
|
||||
class llama_batch_allocr;
|
||||
class llama_moe_cache;
|
||||
|
||||
class llama_io_read_i;
|
||||
class llama_io_write_i;
|
||||
@@ -271,7 +273,7 @@ private:
|
||||
|
||||
llm_graph_cb graph_get_cb() const;
|
||||
|
||||
// ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID
|
||||
// ggml_backend_sched copy callback, copies only the experts used by MUL_MAT_ID and updates the MoE cache
|
||||
static bool sched_copy_experts(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph, void * user_data);
|
||||
|
||||
// disable auto fused ops (Flash Attention, Gated Delta Net) whose op lands on a device
|
||||
@@ -299,6 +301,7 @@ private:
|
||||
llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably
|
||||
|
||||
llama_memory_ptr memory;
|
||||
llama_moe_cache_ptr moe_cache;
|
||||
|
||||
// decode output (2-dimensional array: [n_outputs][n_vocab])
|
||||
buffer_view<float> logits = {nullptr, 0};
|
||||
|
||||
@@ -55,6 +55,8 @@ struct llama_cparams {
|
||||
bool pipeline_parallel;
|
||||
bool training; // set by llama_opt_init()
|
||||
|
||||
size_t moe_cache_size;
|
||||
|
||||
std::vector<bool> embeddings_layer_inp; // [n_layer()] extract input embeddings for layer
|
||||
|
||||
enum llama_context_type ctx_type;
|
||||
|
||||
+54
-6
@@ -2,6 +2,7 @@
|
||||
|
||||
#include "llama-impl.h"
|
||||
#include "llama-model.h"
|
||||
#include "llama-moe-cache.h"
|
||||
#include "llama-batch.h"
|
||||
#include "llama-cparams.h"
|
||||
#include "llama-sampler.h"
|
||||
@@ -1523,6 +1524,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) :
|
||||
loras (params.loras),
|
||||
mctx (params.mctx),
|
||||
cross (params.cross),
|
||||
moe_cache (params.moe_cache),
|
||||
prec_policy (params.prec_policy),
|
||||
samplers (params.samplers),
|
||||
cb_func (params.cb),
|
||||
@@ -1589,8 +1591,12 @@ ggml_tensor * llm_graph_context::build_lora_mm_id(
|
||||
ggml_tensor * w, // ggml_tensor * as
|
||||
ggml_tensor * cur, // ggml_tensor * b
|
||||
ggml_tensor * ids,
|
||||
ggml_tensor * w_s) const {
|
||||
ggml_tensor * res = ggml_mul_mat_id(ctx0, w, cur, ids);
|
||||
ggml_tensor * w_s,
|
||||
ggml_tensor * slots) const {
|
||||
// the experts in the MoE cache are selected by their slots
|
||||
ggml_tensor * res = slots == nullptr ?
|
||||
ggml_mul_mat_id(ctx0, w, cur, ids) :
|
||||
ggml_mul_mat_id(ctx0, moe_cache->get_experts(w), cur, slots);
|
||||
|
||||
if (prec_policy) {
|
||||
prec_policy->apply(res);
|
||||
@@ -2205,6 +2211,9 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||
//call early so that topk-moe can be used
|
||||
ggml_build_forward_expand(gf, weights);
|
||||
|
||||
// the experts of host-resident layers may be read from the MoE cache
|
||||
ggml_tensor * slots = build_moe_cache_slots(selected_experts, up_exps, gate_exps, down_exps, gate_up_exps, il);
|
||||
|
||||
cur = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
|
||||
|
||||
if (weight_before_ffn) {
|
||||
@@ -2219,7 +2228,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||
|
||||
if (gate_up_exps) {
|
||||
// merged gate_up path: one mul_mat_id, then split into gate and up views
|
||||
ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s); // [n_ff*2, n_expert_used, n_tokens]
|
||||
ggml_tensor * gate_up = build_lora_mm_id(gate_up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff*2, n_expert_used, n_tokens]
|
||||
cb(gate_up, "ffn_moe_gate_up", il);
|
||||
|
||||
if (up_exps_s) {
|
||||
@@ -2238,7 +2247,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||
cb(up, "ffn_moe_up", il);
|
||||
} else {
|
||||
// separate gate and up path
|
||||
up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s); // [n_ff, n_expert_used, n_tokens]
|
||||
up = build_lora_mm_id(up_exps, cur, selected_experts, up_exps_s, slots); // [n_ff, n_expert_used, n_tokens]
|
||||
cb(up, "ffn_moe_up", il);
|
||||
|
||||
if (up_exps_s) {
|
||||
@@ -2251,7 +2260,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||
}
|
||||
|
||||
if (gate_exps) {
|
||||
cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s); // [n_ff, n_expert_used, n_tokens]
|
||||
cur = build_lora_mm_id(gate_exps, cur, selected_experts, gate_exps_s, slots); // [n_ff, n_expert_used, n_tokens]
|
||||
cb(cur, "ffn_moe_gate", il);
|
||||
} else {
|
||||
cur = up;
|
||||
@@ -2352,7 +2361,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||
GGML_ABORT("fatal error");
|
||||
}
|
||||
|
||||
experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s); // [n_embd, n_expert_used, n_tokens]
|
||||
experts = build_lora_mm_id(down_exps, cur, selected_experts, down_exps_s, slots); // [n_embd, n_expert_used, n_tokens]
|
||||
if (arch == LLM_ARCH_MISTRAL4) {
|
||||
// src1 can exceed F16 range
|
||||
ggml_prec_set_src(experts, GGML_PREC_F32, 1);
|
||||
@@ -2409,6 +2418,45 @@ ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||
return moe_out;
|
||||
}
|
||||
|
||||
ggml_tensor * llm_graph_context::build_moe_cache_slots(
|
||||
ggml_tensor * selected_experts,
|
||||
ggml_tensor * up_exps,
|
||||
ggml_tensor * gate_exps,
|
||||
ggml_tensor * down_exps,
|
||||
ggml_tensor * gate_up_exps,
|
||||
int il) const {
|
||||
if (moe_cache == nullptr) {
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ggml_tensor * slot_map = moe_cache->get_slot_map(il, selected_experts->ne[1], selected_experts->ne[0]);
|
||||
if (slot_map == nullptr) {
|
||||
return nullptr;
|
||||
}
|
||||
for (ggml_tensor * w : { up_exps, gate_exps, down_exps, gate_up_exps }) {
|
||||
if (w != nullptr && moe_cache->get_experts(w) == nullptr) {
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
ggml_tensor * ids = selected_experts;
|
||||
if (!ggml_is_contiguous(ids)) {
|
||||
ids = ggml_cont(ctx0, ids);
|
||||
}
|
||||
ids = ggml_reshape_1d(ctx0, ids, ggml_nelements(ids));
|
||||
|
||||
// the slot map is a host weight, so the scheduler starts a new split here and copies it with the copy callback
|
||||
// the callback reads the selected experts, uploads the missing ones and updates the slot map
|
||||
ggml_tensor * slots = ggml_get_rows(ctx0, slot_map, ids); // [1, n_expert_used*n_tokens]
|
||||
if (!ggml_backend_supports_op(moe_cache->backend(), slots)) {
|
||||
return nullptr;
|
||||
}
|
||||
ggml_backend_sched_set_tensor_backend(sched, slots, moe_cache->backend());
|
||||
cb(slots, "ffn_moe_slots", il);
|
||||
|
||||
return ggml_reshape_2d(ctx0, slots, selected_experts->ne[0], selected_experts->ne[1]); // [n_expert_used, n_tokens]
|
||||
}
|
||||
|
||||
// input embeddings with optional lora
|
||||
ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float tok_scale) const {
|
||||
const int64_t n_embd_inp = hparams.n_embd_inp();
|
||||
|
||||
+16
-1
@@ -21,6 +21,8 @@ struct llama_cparams;
|
||||
struct llama_layer;
|
||||
struct llama_prec_policy;
|
||||
|
||||
class llama_moe_cache;
|
||||
|
||||
struct llama_memory_context_i;
|
||||
|
||||
class llama_kv_cache_context;
|
||||
@@ -793,6 +795,7 @@ struct llm_graph_params {
|
||||
const llama_adapter_loras * loras;
|
||||
const llama_memory_context_i * mctx;
|
||||
const llama_cross * cross;
|
||||
const llama_moe_cache * moe_cache;
|
||||
|
||||
const llama_prec_policy * prec_policy = nullptr;
|
||||
|
||||
@@ -1036,6 +1039,7 @@ struct llm_graph_context {
|
||||
const llama_adapter_loras * loras;
|
||||
const llama_memory_context_i * mctx;
|
||||
const llama_cross * cross;
|
||||
const llama_moe_cache * moe_cache;
|
||||
|
||||
const llama_prec_policy * prec_policy;
|
||||
|
||||
@@ -1078,11 +1082,13 @@ struct llm_graph_context {
|
||||
ggml_tensor * w_s = nullptr) const;
|
||||
|
||||
// do mat_mul_id, while optionally apply lora and per-expert scale
|
||||
// if slots is set, the experts are read from the MoE cache at these slots (see build_moe_cache_slots)
|
||||
ggml_tensor * build_lora_mm_id(
|
||||
ggml_tensor * w, // ggml_tensor * as
|
||||
ggml_tensor * cur, // ggml_tensor * b
|
||||
ggml_tensor * ids,
|
||||
ggml_tensor * w_s = nullptr) const;
|
||||
ggml_tensor * w_s = nullptr,
|
||||
ggml_tensor * slots = nullptr) const;
|
||||
|
||||
ggml_tensor * build_norm(
|
||||
ggml_tensor * cur,
|
||||
@@ -1179,6 +1185,15 @@ struct llm_graph_context {
|
||||
ggml_tensor * down_exps_s = nullptr,
|
||||
ggml_tensor * selected_experts_in = nullptr) const;
|
||||
|
||||
// the slots of the selected experts in the MoE cache, nullptr if the experts of the layer are not read from the cache
|
||||
ggml_tensor * build_moe_cache_slots(
|
||||
ggml_tensor * selected_experts,
|
||||
ggml_tensor * up_exps,
|
||||
ggml_tensor * gate_exps,
|
||||
ggml_tensor * down_exps,
|
||||
ggml_tensor * gate_up_exps,
|
||||
int il) const;
|
||||
|
||||
//
|
||||
// inputs
|
||||
//
|
||||
|
||||
@@ -0,0 +1,550 @@
|
||||
#include "llama-moe-cache.h"
|
||||
|
||||
#include "llama-impl.h"
|
||||
#include "llama-model.h"
|
||||
|
||||
#include "ggml-cpp.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <stdexcept>
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
namespace {
|
||||
|
||||
// LRU of the experts of a group of layers, the slot of each expert is kept in the slot map of its layer
|
||||
struct moe_cache_lru {
|
||||
int32_t n_expert = 0;
|
||||
int32_t n_slots = 0;
|
||||
|
||||
std::vector<int32_t *> slot_map; // [n_layer] data of the slot maps, -1 if the expert is not cached
|
||||
std::vector<int32_t> key_of; // [n_slots] il*n_expert + expert, -1 if empty
|
||||
|
||||
// doubly linked list of the slots, head is the least recently used
|
||||
std::vector<int32_t> prev;
|
||||
std::vector<int32_t> next;
|
||||
int32_t head = -1;
|
||||
int32_t tail = -1;
|
||||
|
||||
std::vector<uint32_t> seen; // [n_expert]
|
||||
uint32_t seen_gen = 0;
|
||||
std::vector<int32_t> uniq;
|
||||
|
||||
void init(int32_t n_layer, int32_t n_expert, int32_t n_slots) {
|
||||
this->n_expert = n_expert;
|
||||
this->n_slots = n_slots;
|
||||
slot_map.assign(n_layer, nullptr);
|
||||
key_of.assign(n_slots, -1);
|
||||
prev.resize(n_slots);
|
||||
next.resize(n_slots);
|
||||
for (int32_t s = 0; s < n_slots; ++s) {
|
||||
prev[s] = s - 1;
|
||||
next[s] = s + 1 < n_slots ? s + 1 : -1;
|
||||
}
|
||||
head = 0;
|
||||
tail = n_slots - 1;
|
||||
seen.assign(n_expert, 0);
|
||||
}
|
||||
|
||||
// move slot s to the tail (most recently used)
|
||||
void touch(int32_t s) {
|
||||
if (s == tail) {
|
||||
return;
|
||||
}
|
||||
if (prev[s] >= 0) {
|
||||
next[prev[s]] = next[s];
|
||||
} else {
|
||||
head = next[s];
|
||||
}
|
||||
prev[next[s]] = prev[s];
|
||||
|
||||
prev[s] = tail;
|
||||
next[s] = -1;
|
||||
next[tail] = s;
|
||||
tail = s;
|
||||
}
|
||||
|
||||
struct fill {
|
||||
int32_t expert;
|
||||
int32_t slot;
|
||||
};
|
||||
|
||||
// give a slot to each expert selected by ids in layer il, the misses evict the least recently used experts
|
||||
// returns false if the ids select more distinct experts than there are slots
|
||||
bool plan(int32_t il, const int32_t * ids, size_t n_ids, std::vector<fill> & fills, size_t & n_hit) {
|
||||
fills.clear();
|
||||
n_hit = 0;
|
||||
|
||||
if (++seen_gen == 0) {
|
||||
std::fill(seen.begin(), seen.end(), 0);
|
||||
seen_gen = 1;
|
||||
}
|
||||
uniq.clear();
|
||||
for (size_t i = 0; i < n_ids; ++i) {
|
||||
GGML_ASSERT(ids[i] >= 0 && ids[i] < n_expert);
|
||||
if (seen[ids[i]] != seen_gen) {
|
||||
seen[ids[i]] = seen_gen;
|
||||
uniq.push_back(ids[i]);
|
||||
}
|
||||
}
|
||||
if (uniq.size() > (size_t) n_slots) {
|
||||
return false;
|
||||
}
|
||||
|
||||
int32_t * slots = slot_map[il];
|
||||
|
||||
// hits go to the tail first, so the head can be evicted below
|
||||
for (int32_t e : uniq) {
|
||||
if (slots[e] >= 0) {
|
||||
touch(slots[e]);
|
||||
n_hit++;
|
||||
}
|
||||
}
|
||||
// sorted misses usually get consecutive slots, so the uploads can be merged
|
||||
std::sort(uniq.begin(), uniq.end());
|
||||
for (int32_t e : uniq) {
|
||||
if (slots[e] >= 0) {
|
||||
continue;
|
||||
}
|
||||
const int32_t s = head;
|
||||
if (key_of[s] >= 0) {
|
||||
slot_map[key_of[s] / n_expert][key_of[s] % n_expert] = -1;
|
||||
}
|
||||
key_of[s] = il*n_expert + e;
|
||||
slots[e] = s;
|
||||
touch(s);
|
||||
fills.push_back({ e, s });
|
||||
}
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
// gate, up, down or gate_up, down
|
||||
static std::vector<ggml_tensor *> llama_moe_cache_layer_experts(const llama_layer & layer) {
|
||||
std::vector<ggml_tensor *> res;
|
||||
for (ggml_tensor * t : { layer.ffn_gate_up_exps, layer.ffn_gate_exps, layer.ffn_up_exps, layer.ffn_down_exps }) {
|
||||
if (t != nullptr) {
|
||||
res.push_back(t);
|
||||
}
|
||||
}
|
||||
return res;
|
||||
}
|
||||
|
||||
static bool llama_moe_cache_same_layout(const std::vector<ggml_tensor *> & a, const std::vector<ggml_tensor *> & b) {
|
||||
if (a.size() != b.size()) {
|
||||
return false;
|
||||
}
|
||||
for (size_t i = 0; i < a.size(); ++i) {
|
||||
if (a[i]->type != b[i]->type || !ggml_are_same_shape(a[i], b[i]) || a[i]->nb[2] != b[i]->nb[2]) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool llama_moe_cache_is_host_weight(const ggml_tensor * t) {
|
||||
return t->buffer != nullptr &&
|
||||
ggml_backend_buffer_get_usage(t->buffer) == GGML_BACKEND_BUFFER_USAGE_WEIGHTS &&
|
||||
ggml_backend_buffer_is_host(t->buffer);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
struct llama_moe_cache::impl {
|
||||
// layers with the same expert tensor layout share the banks and the LRU of a group
|
||||
struct group {
|
||||
std::vector<ggml_tensor *> ref; // expert tensors of the first layer
|
||||
std::vector<int32_t> layers;
|
||||
std::vector<ggml_tensor *> banks; // device storage of all slots, one per expert tensor
|
||||
size_t host_bytes = 0;
|
||||
int32_t n_slots = 0;
|
||||
moe_cache_lru lru;
|
||||
};
|
||||
|
||||
struct layer {
|
||||
int32_t ig = -1; // -1 if the layer is not cached
|
||||
ggml_tensor * slot_map = nullptr; // I32 [1, n_expert] in host memory
|
||||
std::vector<ggml_tensor *> experts; // host expert tensors, in the order of the banks
|
||||
};
|
||||
|
||||
struct binding {
|
||||
int32_t il;
|
||||
int32_t ip; // index of the bank
|
||||
ggml_tensor * cached; // view of the bank used in place of the host experts
|
||||
};
|
||||
|
||||
struct stats {
|
||||
size_t hits = 0;
|
||||
size_t misses = 0;
|
||||
size_t bytes = 0;
|
||||
};
|
||||
|
||||
static constexpr int64_t max_batch = 32;
|
||||
|
||||
ggml_backend_t backend;
|
||||
int32_t n_expert_used;
|
||||
|
||||
stats stats_small; // up to 8 tokens per ubatch
|
||||
stats stats_large;
|
||||
stats stats_copy; // experts copied from the cache for large batches
|
||||
|
||||
std::vector<group> groups;
|
||||
std::vector<layer> layers;
|
||||
std::unordered_map<const ggml_tensor *, binding> bindings; // host experts -> cached experts
|
||||
std::unordered_map<const ggml_tensor *, int32_t> layer_of; // slot map -> layer
|
||||
|
||||
std::vector<int32_t> ids;
|
||||
std::vector<moe_cache_lru::fill> fills;
|
||||
|
||||
// banks and their views on the device
|
||||
ggml_context_ptr ctx;
|
||||
ggml_backend_buffer_ptr buf;
|
||||
size_t buf_size = 0;
|
||||
|
||||
// slot maps in host memory
|
||||
ggml_context_ptr ctx_host;
|
||||
ggml_backend_buffer_ptr buf_host;
|
||||
size_t buf_host_size = 0;
|
||||
|
||||
// views used by copy_experts
|
||||
ggml_context_ptr ctx_views;
|
||||
|
||||
impl(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
|
||||
backend(backend), n_expert_used(model.hparams.n_expert_used_max()), layers(model.layers.size()) {
|
||||
ggml_backend_dev_t dev = ggml_backend_get_device(backend);
|
||||
const auto dev_type = ggml_backend_dev_type(dev);
|
||||
if (dev_type != GGML_BACKEND_DEVICE_TYPE_GPU && dev_type != GGML_BACKEND_DEVICE_TYPE_IGPU) {
|
||||
throw std::runtime_error("MoE cache requires a GPU backend");
|
||||
}
|
||||
if (model.split_mode() == LLAMA_SPLIT_MODE_TENSOR) {
|
||||
throw std::runtime_error("MoE cache does not support tensor parallelism");
|
||||
}
|
||||
if (model.hparams.n_expert == 0 || n_expert_used == 0) {
|
||||
throw std::runtime_error("MoE cache requires a MoE model");
|
||||
}
|
||||
|
||||
// only cache layers that keep all of their experts in host memory
|
||||
size_t host_bytes = 0;
|
||||
for (size_t il = 0; il < model.layers.size(); ++il) {
|
||||
auto experts = llama_moe_cache_layer_experts(model.layers[il]);
|
||||
if (experts.empty() || model.dev_layer(il) != dev ||
|
||||
!std::all_of(experts.begin(), experts.end(), llama_moe_cache_is_host_weight)) {
|
||||
continue;
|
||||
}
|
||||
auto it = std::find_if(groups.begin(), groups.end(), [&](const group & g) { return llama_moe_cache_same_layout(g.ref, experts); });
|
||||
if (it == groups.end()) {
|
||||
groups.emplace_back();
|
||||
it = groups.end() - 1;
|
||||
it->ref = experts;
|
||||
}
|
||||
it->layers.push_back(il);
|
||||
for (const ggml_tensor * t : experts) {
|
||||
it->host_bytes += ggml_nbytes(t);
|
||||
host_bytes += ggml_nbytes(t);
|
||||
}
|
||||
}
|
||||
if (groups.empty()) {
|
||||
LLAMA_LOG_WARN("%s: no layer has all of its experts in host memory, MoE cache is disabled\n", __func__);
|
||||
return;
|
||||
}
|
||||
|
||||
// one extra slot at the end, CUDA MMQ can read past the last expert
|
||||
const size_t alignment = ggml_backend_buft_get_alignment(buft);
|
||||
auto alloc_size = [&](const group & g, int32_t n_slots) {
|
||||
size_t res = 0;
|
||||
for (const ggml_tensor * t : g.ref) {
|
||||
res += GGML_PAD(t->nb[2]*(n_slots + 1), alignment);
|
||||
}
|
||||
return res;
|
||||
};
|
||||
|
||||
// split the budget by the size of the experts, so each group caches the same fraction of its experts
|
||||
size_t n_tensors = 0;
|
||||
size_t n_tensors_host = 0;
|
||||
for (group & g : groups) {
|
||||
const int32_t n_expert = g.ref[0]->ne[2];
|
||||
const size_t budget = (size_t) ((double) size*g.host_bytes/host_bytes);
|
||||
const int32_t max_slots = g.layers.size()*n_expert;
|
||||
while (g.n_slots < max_slots && alloc_size(g, g.n_slots + 1) <= budget) {
|
||||
g.n_slots++;
|
||||
}
|
||||
if (g.n_slots < n_expert_used) {
|
||||
LLAMA_LOG_WARN("%s: MoE cache budget is too small for %zu layers, they are not cached\n", __func__, g.layers.size());
|
||||
g.n_slots = 0;
|
||||
continue;
|
||||
}
|
||||
g.lru.init(model.layers.size(), n_expert, g.n_slots);
|
||||
n_tensors += g.ref.size()*(1 + g.layers.size());
|
||||
n_tensors_host += g.layers.size();
|
||||
}
|
||||
if (n_tensors == 0) {
|
||||
throw std::runtime_error("MoE cache is too small to hold the experts of one token");
|
||||
}
|
||||
|
||||
auto init_ctx = [](size_t n_tensors) {
|
||||
ggml_init_params params = {
|
||||
/*.mem_size =*/ n_tensors*ggml_tensor_overhead(),
|
||||
/*.mem_buffer =*/ nullptr,
|
||||
/*.no_alloc =*/ true,
|
||||
};
|
||||
ggml_context_ptr res(ggml_init(params));
|
||||
if (!res) {
|
||||
throw std::runtime_error("failed to create the MoE cache context");
|
||||
}
|
||||
return res;
|
||||
};
|
||||
ctx = init_ctx(n_tensors);
|
||||
ctx_host = init_ctx(n_tensors_host);
|
||||
ctx_views = init_ctx(2);
|
||||
|
||||
ggml_backend_buffer_type_t buft_host = ggml_backend_cpu_buffer_type();
|
||||
const size_t alignment_host = ggml_backend_buft_get_alignment(buft_host);
|
||||
|
||||
for (size_t ig = 0; ig < groups.size(); ++ig) {
|
||||
group & g = groups[ig];
|
||||
if (g.n_slots == 0) {
|
||||
continue;
|
||||
}
|
||||
for (const ggml_tensor * t : g.ref) {
|
||||
ggml_tensor * bank = ggml_new_tensor_3d(ctx.get(), t->type, t->ne[0], t->ne[1], g.n_slots + 1);
|
||||
GGML_ASSERT(bank->nb[2] == t->nb[2]);
|
||||
ggml_format_name(bank, "moe_cache.%zu.%s", ig, t->name);
|
||||
g.banks.push_back(bank);
|
||||
}
|
||||
for (int32_t il : g.layers) {
|
||||
layer & l = layers[il];
|
||||
l.ig = (int32_t) ig;
|
||||
l.experts = llama_moe_cache_layer_experts(model.layers[il]);
|
||||
for (size_t ip = 0; ip < l.experts.size(); ++ip) {
|
||||
ggml_tensor * bank = g.banks[ip];
|
||||
ggml_tensor * cached = ggml_view_3d(ctx.get(), bank, bank->ne[0], bank->ne[1], g.n_slots, bank->nb[1], bank->nb[2], 0);
|
||||
ggml_format_name(cached, "moe_cache.%s", l.experts[ip]->name);
|
||||
bindings[l.experts[ip]] = { il, (int32_t) ip, cached };
|
||||
}
|
||||
l.slot_map = ggml_new_tensor_2d(ctx_host.get(), GGML_TYPE_I32, 1, g.ref[0]->ne[2]);
|
||||
ggml_format_name(l.slot_map, "moe_cache.slot_map-%d", il);
|
||||
layer_of[l.slot_map] = il;
|
||||
buf_host_size += GGML_PAD(ggml_nbytes(l.slot_map), alignment_host);
|
||||
}
|
||||
buf_size += alloc_size(g, g.n_slots);
|
||||
}
|
||||
|
||||
if (model.hparams.no_alloc) {
|
||||
// only used to measure the memory use, see llama_context::memory_breakdown
|
||||
buf.reset(ggml_backend_buft_alloc_buffer(buft, 0));
|
||||
buf_host.reset(ggml_backend_buft_alloc_buffer(buft_host, 0));
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx.get()); t != nullptr; t = ggml_get_next_tensor(ctx.get(), t)) {
|
||||
t->buffer = buf.get();
|
||||
}
|
||||
for (ggml_tensor * t = ggml_get_first_tensor(ctx_host.get()); t != nullptr; t = ggml_get_next_tensor(ctx_host.get(), t)) {
|
||||
t->buffer = buf_host.get();
|
||||
}
|
||||
} else {
|
||||
buf.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx.get(), buft));
|
||||
buf_host.reset(ggml_backend_alloc_ctx_tensors_from_buft(ctx_host.get(), buft_host));
|
||||
if (!buf || !buf_host) {
|
||||
throw std::runtime_error("failed to allocate the MoE cache buffers");
|
||||
}
|
||||
ggml_backend_buffer_clear(buf.get(), 0);
|
||||
ggml_backend_buffer_clear(buf_host.get(), 0xff); // all slots are -1
|
||||
buf_size = ggml_backend_buffer_get_size(buf.get());
|
||||
buf_host_size = ggml_backend_buffer_get_size(buf_host.get());
|
||||
|
||||
for (group & g : groups) {
|
||||
for (int32_t il : g.layers) {
|
||||
if (layers[il].slot_map != nullptr) {
|
||||
g.lru.slot_map[il] = (int32_t *) layers[il].slot_map->data;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// as weights, the ops that read the banks run on the device and the slot maps are copied with the copy callback
|
||||
ggml_backend_buffer_set_usage(buf.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
|
||||
ggml_backend_buffer_set_usage(buf_host.get(), GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
|
||||
|
||||
LLAMA_LOG_INFO("%s: %10s MoE cache size = %8.2f MiB for %.2f MiB of host experts\n", __func__,
|
||||
ggml_backend_buft_name(buft), buf_size/1024.0/1024.0, host_bytes/1024.0/1024.0);
|
||||
for (const group & g : groups) {
|
||||
LLAMA_LOG_INFO("%s: %2zu layers, %s: %5d slots (%.1f%%)\n", __func__,
|
||||
g.layers.size(), ggml_type_name(g.ref.back()->type), g.n_slots, 100.0*g.n_slots/(g.layers.size()*g.ref[0]->ne[2]));
|
||||
}
|
||||
}
|
||||
|
||||
~impl() {
|
||||
log_stats();
|
||||
}
|
||||
|
||||
ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
|
||||
if (il < 0 || il >= (int32_t) layers.size() || layers[il].ig < 0) {
|
||||
return nullptr;
|
||||
}
|
||||
const layer & l = layers[il];
|
||||
|
||||
// large batches use most experts of a layer, so they gain little from the cache and would evict the experts used in generation
|
||||
if (n_tokens == 0 || n_tokens > max_batch || std::min(n_tokens*n_expert_used, l.slot_map->ne[1]) > groups[l.ig].n_slots) {
|
||||
return nullptr;
|
||||
}
|
||||
return l.slot_map;
|
||||
}
|
||||
|
||||
ggml_tensor * get_experts(const ggml_tensor * w) const {
|
||||
const auto it = bindings.find(w);
|
||||
return it != bindings.end() ? it->second.cached : nullptr;
|
||||
}
|
||||
|
||||
int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
|
||||
const auto it = bindings.find(w);
|
||||
if (it == bindings.end() || backend != this->backend) {
|
||||
return 0;
|
||||
}
|
||||
const binding & b = it->second;
|
||||
const group & g = groups[layers[b.il].ig];
|
||||
|
||||
// large batches only read the cache, so the experts used in generation stay in it
|
||||
const int32_t * slots = g.lru.slot_map[b.il];
|
||||
if (slots == nullptr || slots[e] < 0) {
|
||||
return 0;
|
||||
}
|
||||
int64_t n = 1;
|
||||
while (e + n <= last && slots[e + n] == slots[e] + n) {
|
||||
n++;
|
||||
}
|
||||
|
||||
ggml_tensor * bank = g.banks[b.ip];
|
||||
ggml_reset(ctx_views.get());
|
||||
ggml_tensor * src_view = ggml_view_3d(ctx_views.get(), bank, bank->ne[0], bank->ne[1], n, bank->nb[1], bank->nb[2], slots[e]*bank->nb[2]);
|
||||
ggml_tensor * dst_view = ggml_view_3d(ctx_views.get(), dst, dst->ne[0], dst->ne[1], n, dst->nb[1], dst->nb[2], e*dst->nb[2]);
|
||||
ggml_backend_view_init(src_view);
|
||||
ggml_backend_view_init(dst_view);
|
||||
ggml_backend_tensor_copy_async(backend, backend, src_view, dst_view);
|
||||
|
||||
stats_copy.hits += n;
|
||||
stats_copy.bytes += ggml_nbytes(src_view);
|
||||
|
||||
return n;
|
||||
}
|
||||
|
||||
bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) {
|
||||
const auto it = layer_of.find(src);
|
||||
if (it == layer_of.end()) {
|
||||
return false;
|
||||
}
|
||||
const int32_t il = it->second;
|
||||
const layer & l = layers[il];
|
||||
group & g = groups[l.ig];
|
||||
|
||||
GGML_ASSERT(backend == this->backend);
|
||||
|
||||
// the get_rows that looks up the slots of the selected experts
|
||||
const int n_nodes = ggml_graph_n_nodes(graph);
|
||||
const ggml_tensor * lookup = nullptr;
|
||||
for (int i = 0; i < n_nodes && lookup == nullptr; ++i) {
|
||||
const ggml_tensor * node = ggml_graph_node(graph, i);
|
||||
if (node->op == GGML_OP_GET_ROWS && node->src[0] == dst) {
|
||||
lookup = node;
|
||||
}
|
||||
}
|
||||
GGML_ASSERT(lookup != nullptr);
|
||||
|
||||
// the selected experts must be computed in an earlier split
|
||||
// the scheduler starts a new split at the lookup because it reads a host weight, but only if the split already has inputs
|
||||
const ggml_tensor * sel = lookup->src[1];
|
||||
for (int i = 0; i < n_nodes; ++i) {
|
||||
const ggml_tensor * node = ggml_graph_node(graph, i);
|
||||
if (node == sel || node == sel->view_src) {
|
||||
GGML_ABORT("the experts of layer %d are selected in the same split as their MoE cache lookup", il);
|
||||
}
|
||||
}
|
||||
GGML_ASSERT(ggml_is_contiguous(sel));
|
||||
|
||||
ids.resize(ggml_nelements(sel));
|
||||
ggml_backend_tensor_get_async(backend, sel, ids.data(), 0, ggml_nbytes(sel));
|
||||
ggml_backend_synchronize(backend);
|
||||
|
||||
size_t n_hit = 0;
|
||||
if (!g.lru.plan(il, ids.data(), ids.size(), fills, n_hit)) {
|
||||
GGML_ABORT("the MoE cache is too small for the experts selected in layer %d", il);
|
||||
}
|
||||
|
||||
// upload the missing experts, consecutive experts going to consecutive slots are uploaded together
|
||||
size_t bytes = 0;
|
||||
for (size_t ip = 0; ip < l.experts.size(); ++ip) {
|
||||
const ggml_tensor * w = l.experts[ip];
|
||||
ggml_tensor * bank = g.banks[ip];
|
||||
const size_t expert_size = w->nb[2];
|
||||
for (size_t i = 0; i < fills.size();) {
|
||||
size_t n = 1;
|
||||
while (i + n < fills.size() && fills[i + n].expert == fills[i].expert + (int32_t) n && fills[i + n].slot == fills[i].slot + (int32_t) n) {
|
||||
n++;
|
||||
}
|
||||
ggml_backend_tensor_set_async(backend, bank, (const uint8_t *) w->data + fills[i].expert*expert_size, fills[i].slot*expert_size, n*expert_size);
|
||||
bytes += n*expert_size;
|
||||
i += n;
|
||||
}
|
||||
}
|
||||
|
||||
stats & st = ids.size() <= (size_t) 8*n_expert_used ? stats_small : stats_large;
|
||||
st.hits += n_hit;
|
||||
st.misses += fills.size();
|
||||
st.bytes += bytes;
|
||||
|
||||
// the next copy synchronizes the backend before it changes the slot map again
|
||||
ggml_backend_tensor_set_async(backend, dst, src->data, 0, ggml_nbytes(src));
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void log_stats() const {
|
||||
auto log = [](const char * name, const stats & st) {
|
||||
const size_t n = st.hits + st.misses;
|
||||
if (n == 0) {
|
||||
return;
|
||||
}
|
||||
LLAMA_LOG_INFO("llama_moe_cache: %s: hits = %zu, misses = %zu, hit rate = %.2f%%, uploaded = %.2f MiB\n",
|
||||
name, st.hits, st.misses, 100.0*st.hits/n, st.bytes/1024.0/1024.0);
|
||||
};
|
||||
log("ubatch <= 8", stats_small);
|
||||
log("ubatch > 8", stats_large);
|
||||
if (stats_copy.hits > 0) {
|
||||
LLAMA_LOG_INFO("llama_moe_cache: large batches: %zu experts copied from the cache, %.2f MiB\n", stats_copy.hits, stats_copy.bytes/1024.0/1024.0);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
llama_moe_cache::llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size) :
|
||||
pimpl(new impl(model, backend, buft, size)) {
|
||||
}
|
||||
|
||||
llama_moe_cache::~llama_moe_cache() = default;
|
||||
|
||||
ggml_backend_t llama_moe_cache::backend() const {
|
||||
return pimpl->backend;
|
||||
}
|
||||
|
||||
ggml_tensor * llama_moe_cache::get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const {
|
||||
return pimpl->get_slot_map(il, n_tokens, n_expert_used);
|
||||
}
|
||||
|
||||
ggml_tensor * llama_moe_cache::get_experts(const ggml_tensor * w) const {
|
||||
return pimpl->get_experts(w);
|
||||
}
|
||||
|
||||
bool llama_moe_cache::copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph) {
|
||||
return pimpl->copy(backend, src, dst, graph);
|
||||
}
|
||||
|
||||
int64_t llama_moe_cache::copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last) {
|
||||
return pimpl->copy_experts(backend, w, dst, e, last);
|
||||
}
|
||||
|
||||
std::map<ggml_backend_buffer_type_t, size_t> llama_moe_cache::memory_breakdown() const {
|
||||
std::map<ggml_backend_buffer_type_t, size_t> res;
|
||||
if (pimpl->buf) {
|
||||
res[ggml_backend_buffer_get_type(pimpl->buf.get())] += pimpl->buf_size;
|
||||
}
|
||||
if (pimpl->buf_host) {
|
||||
res[ggml_backend_buffer_get_type(pimpl->buf_host.get())] += pimpl->buf_host_size;
|
||||
}
|
||||
return res;
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
#pragma once
|
||||
|
||||
#include "ggml-backend.h"
|
||||
|
||||
#include <map>
|
||||
#include <memory>
|
||||
|
||||
struct llama_model;
|
||||
|
||||
// keeps the most recently used experts of host-resident MoE layers in a device buffer
|
||||
// each layer has a slot map in host memory: when the scheduler copies it to the device, the copy callback uploads the missing experts
|
||||
class llama_moe_cache {
|
||||
public:
|
||||
llama_moe_cache(const llama_model & model, ggml_backend_t backend, ggml_backend_buffer_type_t buft, size_t size);
|
||||
~llama_moe_cache();
|
||||
|
||||
ggml_backend_t backend() const;
|
||||
|
||||
// the slot map of layer il, if its experts can be read from the cache for n_tokens tokens, nullptr otherwise
|
||||
ggml_tensor * get_slot_map(int32_t il, int64_t n_tokens, int64_t n_expert_used) const;
|
||||
|
||||
// the experts of w in the cache, nullptr if w is not cached
|
||||
ggml_tensor * get_experts(const ggml_tensor * w) const;
|
||||
|
||||
// ggml_backend_sched copy callback, returns false if src is not a slot map
|
||||
bool copy(ggml_backend_t backend, const ggml_tensor * src, ggml_tensor * dst, ggml_cgraph * graph);
|
||||
|
||||
// for large batches: copy the experts of w that are in the cache, starting at expert e and up to expert last, to the copy dst of w
|
||||
// returns the number of experts copied, 0 if expert e is not in the cache
|
||||
int64_t copy_experts(ggml_backend_t backend, const ggml_tensor * w, ggml_tensor * dst, int64_t e, int64_t last);
|
||||
|
||||
std::map<ggml_backend_buffer_type_t, size_t> memory_breakdown() const;
|
||||
|
||||
private:
|
||||
struct impl;
|
||||
std::unique_ptr<impl> pimpl;
|
||||
};
|
||||
|
||||
using llama_moe_cache_ptr = std::unique_ptr<llama_moe_cache>;
|
||||
@@ -495,7 +495,7 @@ static std::pair<llama_model_ptr, llama_context_ptr> get_model_and_ctx(
|
||||
struct gguf_context * gguf_ctx, FILE * file, const size_t seed, const float stdev,
|
||||
const std::vector<ggml_backend_dev_t> & devs,
|
||||
const llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER, bool encode = false,
|
||||
const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr) {
|
||||
const llama_model_tensor_buft_override * tensor_buft_overrides = nullptr, const size_t moe_cache_size = 0) {
|
||||
GGML_ASSERT((gguf_ctx == nullptr) != (file == nullptr));
|
||||
llama_model_params model_params = llama_model_default_params();
|
||||
model_params.progress_callback = silent_model_load_progress;
|
||||
@@ -512,6 +512,11 @@ static std::pair<llama_model_ptr, llama_context_ptr> get_model_and_ctx(
|
||||
if (!encode) {
|
||||
ctx_params.n_ubatch = 64;
|
||||
}
|
||||
if (moe_cache_size > 0) {
|
||||
// the MoE cache is only used for small ubatches
|
||||
ctx_params.moe_cache_size = moe_cache_size;
|
||||
ctx_params.n_ubatch = 2;
|
||||
}
|
||||
|
||||
tensor_data_params tensor_params = { seed, stdev };
|
||||
llama_model_ptr model(gguf_ctx != nullptr ?
|
||||
@@ -865,9 +870,10 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
|
||||
std::string label;
|
||||
llama_split_mode split_mode;
|
||||
bool host_experts; // keep the experts in host memory, see host_experts_test
|
||||
size_t moe_cache_size;
|
||||
|
||||
device_config(std::vector<ggml_backend_dev_t> devs, std::string name, llama_split_mode split_mode, bool host_experts = false)
|
||||
: devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts) {}
|
||||
device_config(std::vector<ggml_backend_dev_t> devs, std::string name, llama_split_mode split_mode, bool host_experts = false, size_t moe_cache_size = 0)
|
||||
: devs(std::move(devs)), label(std::move(name)), split_mode(split_mode), host_experts(host_experts), moe_cache_size(moe_cache_size) {}
|
||||
};
|
||||
|
||||
const llama_model_tensor_buft_override host_experts_overrides[] = {
|
||||
@@ -905,6 +911,16 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
|
||||
dev_configs.emplace_back(devices_meta, "Host experts", LLAMA_SPLIT_MODE_LAYER, true);
|
||||
max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
|
||||
}
|
||||
|
||||
// the ops that use the host experts run on a GPU and read the experts from a cache
|
||||
// the cache has only a few slots (4 for 288 KiB experts), so the experts are evicted and uploaded again
|
||||
if (!devices_meta.empty()) {
|
||||
const enum ggml_backend_dev_type type = ggml_backend_dev_type(devices_meta[0]);
|
||||
if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {
|
||||
dev_configs.emplace_back(std::vector<ggml_backend_dev_t>{devices_meta[0]}, "MoE cache", LLAMA_SPLIT_MODE_LAYER, true, 1536*1024);
|
||||
max_device_label_length = std::max(max_device_label_length, dev_configs.back().label.length());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
size_t max_arch_name_length = 0;
|
||||
@@ -987,7 +1003,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
|
||||
}
|
||||
if (dc.split_mode != LLAMA_SPLIT_MODE_TENSOR || llm_arch_supports_sm_tensor(arch)) {
|
||||
test_executed = true;
|
||||
model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides);
|
||||
model_and_ctx_dev = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size);
|
||||
logits_dev = get_logits(model_and_ctx_dev.first.get(), model_and_ctx_dev.second.get(), tokens, encode);
|
||||
const double nmse_val = nmse(logits_cpu, logits_dev);
|
||||
snprintf(nmse_str, sizeof(nmse_str), "(%.2e)", nmse_val);
|
||||
@@ -1053,7 +1069,7 @@ static int test_backends(const std::string & arch_filter, const size_t seed, con
|
||||
ms.save(file);
|
||||
rewind(file);
|
||||
|
||||
auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides);
|
||||
auto model_and_ctx_roundtrip = get_model_and_ctx(nullptr, file, seed, stdev, dc.devs, dc.split_mode, encode, overrides, dc.moe_cache_size);
|
||||
const std::vector<float> logits_roundtrip = get_logits(
|
||||
model_and_ctx_roundtrip.first.get(), model_and_ctx_roundtrip.second.get(), tokens, encode);
|
||||
status_roundtrip = "\033[1;32mOK\033[0m";
|
||||
|
||||
Reference in New Issue
Block a user