Files
llama-cpp/scripts/quick_bench.sh
T
mozempkandClaude Fable 5 5e68d30d31 swap-stack: duo config — qwen3-4b full GPU, ornith-35b pure CPU
Duo group (resident main+subagent for pi): flip VRAM to the small model.
Ornith experts never touch VRAM; its dense-on-GPU split starved qwen to
176 MiB / 5.8 t/s concurrent. After flip: qwen 43 t/s, ornith 8 t/s.
CUDA_VISIBLE_DEVICES= required for ornith — ngl 0 still allocates ~1GB
pp compute buffer on CUDA builds (OOM+segfault). Duo section in
MOE-FINDINGS.md; also snapshots prior swap-stack migration state.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-10 08:28:32 +02:00

140 lines
4.4 KiB
Bash
Executable File

#!/bin/bash
# Quick benchmark of all models using llama-swap
# Tests current optimized configs from envs/*.env
set -euo pipefail
cd "$(dirname "$0")/.."
SWAP_URL="http://localhost:8089"
RESULTS_DIR="benchmark-results"
TIMESTAMP=$(date +%Y%m%d_%H%M%S)
RESULTS_FILE="$RESULTS_DIR/optimized_${TIMESTAMP}.csv"
mkdir -p "$RESULTS_DIR"
# Check llama-swap is running
if ! curl -sf "$SWAP_URL/status" >/dev/null 2>&1; then
echo "ERROR: llama-swap not running at $SWAP_URL"
echo "Start with: docker compose --profile swap up -d"
exit 1
fi
echo "======================================================================"
echo "OPTIMIZED CONFIG BENCHMARK — $(date)"
echo "======================================================================"
echo ""
echo "Results: $RESULTS_FILE"
echo ""
# CSV header
cat > "$RESULTS_FILE" <<EOF
model,ngl,ctx,threads,batch,ubatch,kv_type,parallel,pp_tokens,tg_tokens,pp_time_s,tg_time_s,pp_t_per_s,tg_t_per_s,status
EOF
# Test one model
bench_model() {
local model_name=$1
echo "=== Testing $model_name ==="
# Switch to model
echo -n " Switching... "
if ! curl -sf -X POST "$SWAP_URL/models/switch" \
-H "Content-Type: application/json" \
-d "{\"model\":\"$model_name\"}" | grep -q "starting\|already_active\|healthy"; then
echo "FAILED"
echo "$model_name,,,,,,,,,,,,,SWITCH_FAILED" >> "$RESULTS_FILE"
return 1
fi
echo "OK"
# Wait for model to load (controller waits internally, but give buffer)
sleep 10
# Check via proxy that llama_server is up
if ! curl -sf "$SWAP_URL/v1/models" >/dev/null 2>&1; then
echo " Model failed to start"
echo "$model_name,,,,,,,,,,,,,START_FAILED" >> "$RESULTS_FILE"
return 1
fi
# Get model metadata from active container
local active_model
active_model=$(curl -sf "$SWAP_URL/status" | grep -oP '"active_model":\s*"\K[^"]+' || echo "unknown")
# Read env file to get params
local env_file="envs/.env.$model_name"
if [ ! -f "$env_file" ]; then
echo " WARNING: $env_file not found"
return 1
fi
source "$env_file"
echo " Config: ngl=$N_GPU_LAYERS ctx=$CTX_SIZE t=$THREADS batch=$BATCH_SIZE/$UBATCH_SIZE kv=$CACHE_TYPE_K parallel=$PARALLEL"
# Prefill test: 512 tokens
echo -n " Prefill (512t)... "
local pp_start pp_end pp_time pp_tps
pp_start=$(date +%s.%N)
local pp_response
pp_response=$(curl -sf -X POST "$SWAP_URL/v1/completions" \
-H "Content-Type: application/json" \
-d '{
"prompt": "'"$(python3 -c "print('The quick brown fox '*64)")"'",
"max_tokens": 1,
"temperature": 0.0
}' 2>&1) || {
echo "FAILED"
echo "$model_name,$N_GPU_LAYERS,$CTX_SIZE,$THREADS,$BATCH_SIZE,$UBATCH_SIZE,$CACHE_TYPE_K,$PARALLEL,512,1,,,,,PP_FAILED" >> "$RESULTS_FILE"
return 1
}
pp_end=$(date +%s.%N)
pp_time=$(echo "$pp_end - $pp_start" | bc -l)
pp_tps=$(echo "scale=1; 512 / $pp_time" | bc -l)
echo "${pp_tps} t/s"
# Generation test: 128 tokens
echo -n " Generation (128t)... "
local tg_start tg_end tg_time tg_tps
tg_start=$(date +%s.%N)
local tg_response
tg_response=$(curl -sf -X POST "$SWAP_URL/v1/completions" \
-H "Content-Type: application/json" \
-d '{
"prompt": "Count from 1 to 100:",
"max_tokens": 128,
"temperature": 0.0
}' 2>&1) || {
echo "FAILED"
echo "$model_name,$N_GPU_LAYERS,$CTX_SIZE,$THREADS,$BATCH_SIZE,$UBATCH_SIZE,$CACHE_TYPE_K,$PARALLEL,512,128,$pp_time,,$pp_tps,,TG_FAILED" >> "$RESULTS_FILE"
return 1
}
tg_end=$(date +%s.%N)
tg_time=$(echo "$tg_end - $tg_start" | bc -l)
tg_tps=$(echo "scale=1; 128 / $tg_time" | bc -l)
echo "${tg_tps} t/s"
# Write results
echo "$model_name,$N_GPU_LAYERS,$CTX_SIZE,$THREADS,$BATCH_SIZE,$UBATCH_SIZE,$CACHE_TYPE_K,$PARALLEL,512,128,$pp_time,$tg_time,$pp_tps,$tg_tps,OK" >> "$RESULTS_FILE"
echo ""
}
# Benchmark all models
for model in ornith-9b qwen35-9b qwen3-4b smollm3-3b gemma4-e2b gemma4-e4b; do
bench_model "$model" || echo " Skipping $model"
done
echo "======================================================================"
echo "SUMMARY"
echo "======================================================================"
echo ""
column -t -s',' "$RESULTS_FILE"
echo ""
echo "Full results: $RESULTS_FILE"