mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-11 07:20:33 +02:00
llama-bench: respect -fitc if bigger than required benchmark size (#28331)
Assisted-By: opencode,llama.cpp,Qwen3.6-35B-A3B,Qwen3.8-27B
This commit is contained in:
@@ -35,7 +35,7 @@ options:
|
||||
--progress print test progress indicators
|
||||
--no-warmup skip warmup runs before benchmarking
|
||||
-fitt, --fit-target <MiB> fit model to device memory with this margin per device in MiB (default: off)
|
||||
-fitc, --fit-ctx <n> minimum ctx size for --fit-target (default: 4096)
|
||||
-fitc, --fit-ctx <n> minimum ctx size for --fit-target (default: 0)
|
||||
-rpc, --rpc <rpc_servers> register RPC devices (comma separated)
|
||||
|
||||
test parameters:
|
||||
|
||||
@@ -441,7 +441,7 @@ static void print_usage(int /* argc */, char ** argv) {
|
||||
printf(" --progress print test progress indicators\n");
|
||||
printf(" --no-warmup skip warmup runs before benchmarking\n");
|
||||
printf(" -fitt, --fit-target <MiB> fit model to device memory with this margin per device in MiB (default: off)\n");
|
||||
printf(" -fitc, --fit-ctx <n> minimum ctx size for --fit-target (default: 4096)\n");
|
||||
printf(" -fitc, --fit-ctx <n> minimum ctx size for --fit-target (default: 0)\n");
|
||||
if (llama_supports_rpc()) {
|
||||
printf(" -rpc, --rpc <rpc_servers> register RPC devices (comma separated)\n");
|
||||
}
|
||||
@@ -2348,7 +2348,8 @@ int llama_bench(int argc, char ** argv) {
|
||||
|
||||
std::vector<size_t> margins(llama_max_devices(), inst.fit_target * 1024 * 1024);
|
||||
|
||||
uint32_t n_ctx_needed = inst.n_prompt + inst.n_gen + inst.n_depth;
|
||||
// fit at least the requested minimum context size, not just the tokens the benchmark processes
|
||||
uint32_t n_ctx_needed = std::max<uint32_t>(inst.n_prompt + inst.n_gen + inst.n_depth, inst.fit_min_ctx);
|
||||
cparams.n_ctx = std::max(cparams.n_ctx, n_ctx_needed);
|
||||
|
||||
common_fit_params(inst.model.c_str(), &mparams, &cparams,
|
||||
|
||||
Reference in New Issue
Block a user