feat(bonsai): add Bonsai-27B-Q1_0 fully-on-GPU (1650 Ti 4GB)
Add a 4th binary to the swap-stack: PrismML llama-server-ternary (Q1_0_g128 ternary kernels, built sm75/CUDA-12.8 static) for Bonsai-27B-Q1_0. Config: 3.8GB weights ALL on the 4GB GPU (ngl 99) + KV in CPU RAM (--no-kv-offload, since VRAM cannot hold both) + q4_0 KV + --flash-attn (GPU attention; iq4_nl forces CPU attention) + tiny batch (-b 64) to fit the ~100MB VRAM headroom. Fits at 3611/3718 MiB. ~7 t/s decode, ~27 pp prefill (1650 Ti has no tensor cores + KV over PCIe = the ceiling). Measured alt configs (all worse): ngl48+KV-VRAM = 8.5 t/s decode but only 10 pp prefill (CPU layers tank prefill); ngl99+KV-VRAM OOMs even at ctx 2048. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -2,16 +2,38 @@
|
||||
# /app/llama-server TurboQuant fork (May 2026) — turbo2/3/4 KV, needed by 9B configs
|
||||
# /app-upstream/llama-server upstream master (Jul 2026) — newest MoE/arch work
|
||||
# /app/llama-server-ik ik_llama.cpp — -ser / -fmoe / -rtr, fast IQ-quant CPU kernels
|
||||
# /app/llama-server-ternary PrismML fork — Q1_0/Q2_0 g128 ternary kernels (Bonsai-27B)
|
||||
# config.yaml picks the binary per model (env: LD_LIBRARY_PATH for upstream).
|
||||
FROM local/llama-cpp-upstream:server-cuda-sm75-mmq AS upstream
|
||||
FROM local/ik-llama:server-cuda-sm75 AS ik
|
||||
|
||||
# ── PrismML ternary build (Q1_0_g128 hybrid-attention kernels, sm75, static) ──
|
||||
# Bonsai-27B-Q1_0: 1-bit ternary, mainline/TheTom lack the g128 kernels. Built
|
||||
# STATIC so it can't collide with the other forks' libs; only needs the CUDA 12.8
|
||||
# runtime libs present in the final turboquant image. USE q4_0 KV + --flash-attn
|
||||
# (NOT iq4_nl — CUDA FA doesn't support iq4_nl → CPU attention). Pinned 7529fdaaf.
|
||||
FROM nvidia/cuda:12.8.0-devel-ubuntu22.04 AS ternary
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
git cmake ninja-build build-essential libcurl4-openssl-dev ca-certificates \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
RUN ln -sf /usr/local/cuda/lib64/stubs/libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1 \
|
||||
&& echo /usr/local/cuda/lib64/stubs > /etc/ld.so.conf.d/cuda-stubs.conf && ldconfig
|
||||
WORKDIR /src
|
||||
RUN git clone https://github.com/PrismML-Eng/llama.cpp.git . \
|
||||
&& git checkout 7529fdaaf && git log --oneline -1
|
||||
RUN cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release \
|
||||
-DGGML_CUDA=ON -DGGML_NATIVE=OFF -DBUILD_SHARED_LIBS=OFF \
|
||||
-DCMAKE_CUDA_ARCHITECTURES=75 -DGGML_CUDA_F16=ON -DLLAMA_CURL=ON \
|
||||
-DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \
|
||||
&& cmake --build build --parallel "$(nproc)" --target llama-server
|
||||
|
||||
FROM local/llama-cpp-turboquant:server-cuda-sm75-mmq
|
||||
|
||||
COPY --from=upstream /app /app-upstream
|
||||
COPY --from=ik /llama-server /app-ik/llama-server
|
||||
COPY --from=ik /usr/local/lib/libllama.so /usr/local/lib/libggml.so /usr/local/lib/libmtmd.so /app-ik/
|
||||
COPY --from=ik /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcudart.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublas.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublasLt.so.12 /app-ik/
|
||||
COPY --from=ternary /src/build/bin/llama-server /app/llama-server-ternary
|
||||
COPY llama-swap /app/llama-swap
|
||||
|
||||
# config.yaml is bind-mounted at runtime (see compose.yaml) so edits
|
||||
|
||||
+35
-194
@@ -3,10 +3,13 @@
|
||||
#
|
||||
# x570-style single-endpoint stack: llama-swap owns :8080, spawns/kills
|
||||
# llama-server per requested `model` field, idle-unloads after TTL.
|
||||
# All per-model tuning migrated verbatim from envs/.env.* (benchmarks
|
||||
# 2026-05-05/06 + MoE work 2026-07-10 — see docs/FINDINGS.md).
|
||||
#
|
||||
# Edit this file → `docker restart llama_swap_stack` (config is bind-mounted).
|
||||
#
|
||||
# CLEANED 2026-07-13: dropped all small/dense models (2B/3B/4B/9B, gemma E2B/E4B,
|
||||
# smollm3, vibethinker), the V-quant probes, and the pi duo group/subagent. Kept
|
||||
# ONLY the 3 decent GPU-offload MoE models. Small models were outclassed by these
|
||||
# MoEs (35B-class knowledge at 3B active) and/or too slow to be useful on 4GB.
|
||||
# ==============================================================================
|
||||
|
||||
healthCheckTimeout: 600 # 35B MoE mmap first-touch can take minutes
|
||||
@@ -36,74 +39,29 @@ macros:
|
||||
--threads 6 --threads-batch 6 -fa on
|
||||
|
||||
"q8-kv": "--cache-type-k q8_0 --cache-type-v q8_0"
|
||||
"q4-kv": "--cache-type-k q4_0 --cache-type-v q4_0"
|
||||
"turbo2-kv": "--cache-type-k turbo2 --cache-type-v turbo2"
|
||||
|
||||
# MoE offload: dense backbone on GPU, routed experts in RAM/page cache.
|
||||
# mmap (no mlock) mandatory for files > RAM.
|
||||
"moe-offload": "--n-gpu-layers 99 --cpu-moe --jinja"
|
||||
|
||||
groups:
|
||||
# Resident duo for pi: main coding model + fast subagent stay loaded together.
|
||||
# VRAM split (2026-07-10): qwen3-4b gets the WHOLE GPU (ngl 99) — ornith's
|
||||
# experts never touched VRAM anyway and its dense-on-GPU split starved qwen
|
||||
# to 176 MiB / 10 t/s. Ornith runs fully CPU (page-cache resident, 3B active).
|
||||
# Threads: main 6 / sub 3 — contention only during overlap.
|
||||
"duo":
|
||||
swap: false
|
||||
exclusive: true
|
||||
members: ["ornith-35b-duo", "qwen3-4b-duo"]
|
||||
|
||||
models:
|
||||
|
||||
# ── pi resident duo ─────────────────────────────────────────────────────────
|
||||
# ── The 3 keepers: GPU-offload MoE (dense backbone on GPU, experts in RAM) ───
|
||||
|
||||
"ornith-35b-duo":
|
||||
name: "Ornith 35B (duo main)"
|
||||
description: "Duo main: fully CPU, 128K ctx, q8 K + turbo2 V (K-side turbo costs 3x CPU decode, V-side is free — probe-verified; full-turbo2 quality gate Δ0.196 upper-bounds this mix). ~880MB KV. CUDA hidden — even ngl 0 tries a ~1GB pp compute buffer"
|
||||
env:
|
||||
- "CUDA_VISIBLE_DEVICES="
|
||||
"ornith-35b":
|
||||
name: "Ornith-1.0-35B Q2_K_L (GPU expert-offload, 128K)"
|
||||
description: "MAIN coder/planner (pi solo). GPU expert-offload (ngl 99, n-cpu-moe 40), on-GPU KV, q8 K + turbo2 V. 28 t/s @ full 128K ctx. ~2.9GB VRAM. Requires nvidia_uvm un-wedged (see memory)."
|
||||
cmd: |
|
||||
${server-base}
|
||||
--cache-type-k q8_0 --cache-type-v turbo2
|
||||
--model /models/deepreinforce-ai_Ornith-1.0-35B-Q2_K_L.gguf
|
||||
--n-gpu-layers 0 --jinja
|
||||
--n-gpu-layers 99 --n-cpu-moe 40 --jinja --reasoning-budget 512
|
||||
--ctx-size 131072
|
||||
--batch-size 1024 --ubatch-size 512
|
||||
--cont-batching --parallel 1
|
||||
--slot-save-path /cache/ornith-duo --cache-reuse 256
|
||||
--slot-save-path /cache/ornith --cache-reuse 256
|
||||
ttl: 0
|
||||
|
||||
"qwen3-4b-duo":
|
||||
name: "Qwen3 4B (duo subagent)"
|
||||
description: "Duo subagent: full GPU (ngl 99), 24K ctx (VRAM ceiling: q4 KV = 40KB/tok, 32K OOMs; turbo KV broken on this model), 3 threads"
|
||||
cmd: |
|
||||
/app/llama-server
|
||||
--host 127.0.0.1 --port ${PORT}
|
||||
--threads 3 --threads-batch 3
|
||||
--flash-attn on ${q4-kv}
|
||||
--model /models/Qwen3-4B-Q4_K_M.gguf
|
||||
--n-gpu-layers 99
|
||||
--ctx-size 24576
|
||||
--batch-size 512 --ubatch-size 256
|
||||
--cont-batching --parallel 1
|
||||
--slot-save-path /cache/qwen-duo --cache-reuse 256
|
||||
ttl: 0
|
||||
|
||||
# ── MoE offload models (2026-07-10) ────────────────────────────────────────
|
||||
|
||||
"gpt-oss-20b":
|
||||
name: "gpt-oss-20b MXFP4"
|
||||
description: "20.9B/3.6B-active MoE, native MXFP4. 16.7 tg / 29.5 pp (n-cpu-moe 21: last 3 expert layers in VRAM, +16%)"
|
||||
cmd: |
|
||||
${server-base} ${q8-kv}
|
||||
--model /models/gpt-oss-20b-mxfp4.gguf
|
||||
--n-gpu-layers 99 --n-cpu-moe 21 --jinja
|
||||
--ctx-size 32768
|
||||
--batch-size 1024 --ubatch-size 512
|
||||
--cont-batching --parallel 1
|
||||
ttl: 300
|
||||
|
||||
"qwen36-35b-q2":
|
||||
name: "Qwen3.6-35B-A3B UD-Q2_K_XL"
|
||||
description: "DAILY DRIVER 35B: 23.0 tg / 36 pp. ds4 asymmetric recipe (dense high-bit, experts 2-bit), 12.3GB page-cache resident, last 3 expert layers in VRAM. Gates pass"
|
||||
@@ -116,154 +74,37 @@ models:
|
||||
--cont-batching --parallel 1
|
||||
ttl: 300
|
||||
|
||||
"ornith-35b":
|
||||
name: "Ornith-1.0-35B Q2_K_L"
|
||||
description: "SPEED KING: 29.1 tg / 66 pp. RL coding finetune (Qwen3.5-MoE arch), 13.1GB page-cache resident, 3 expert layers in VRAM, thinking model. MIT. Gates pass"
|
||||
"gpt-oss-20b":
|
||||
name: "gpt-oss-20b MXFP4"
|
||||
description: "20.9B/3.6B-active MoE, native MXFP4. 16.7 tg / 29.5 pp (n-cpu-moe 21: last 3 expert layers in VRAM, +16%)"
|
||||
cmd: |
|
||||
${server-base} ${q8-kv}
|
||||
--model /models/deepreinforce-ai_Ornith-1.0-35B-Q2_K_L.gguf
|
||||
--n-gpu-layers 99 --n-cpu-moe 37 --jinja
|
||||
--model /models/gpt-oss-20b-mxfp4.gguf
|
||||
--n-gpu-layers 99 --n-cpu-moe 21 --jinja
|
||||
--ctx-size 32768
|
||||
--batch-size 1024 --ubatch-size 512
|
||||
--cont-batching --parallel 1
|
||||
ttl: 300
|
||||
|
||||
# (IQ4 variants of ornith-35b/qwen36-35b deleted 2026-07-10 — 36.5GB disk,
|
||||
# both were >RAM thrash-only until the 64GB upgrade. Re-download if needed:
|
||||
# bartowski Ornith-1.0-35B IQ4_XS, unsloth Qwen3.6-35B-A3B UD-IQ4_XS.)
|
||||
|
||||
# ── Dense 9B (RAM-bandwidth-bound, ~4.4 t/s) ───────────────────────────────
|
||||
# (ik_llama / upstream-master A/B variants removed 2026-07-10 — questions
|
||||
# answered in FINDINGS: ik −23% decode, upstream parity. Macros kept for
|
||||
# future experiments; binaries still in the image.)
|
||||
|
||||
"ornith-9b":
|
||||
name: "Ornith-1.0-9B Q8_0"
|
||||
description: "Coding 9B, MIT. ~4.4 t/s, mlock-pinned"
|
||||
# ── Bonsai-27B Q1_0 (1-bit ternary, PrismML fork) — fully on the 1650 Ti (4GB) ──
|
||||
# 3.8GB weights ALL on GPU (ngl 99); KV in CPU RAM (--no-kv-offload) since VRAM
|
||||
# can't hold both. q4_0 KV + --flash-attn = GPU attention (iq4_nl would force CPU).
|
||||
# Tiny batch (-b 64) keeps the compute buffer under the ~48MB VRAM headroom.
|
||||
# ctx is RAM-limited (KV offloaded), not VRAM. Trade: decode ~11 t/s (KV over
|
||||
# PCIe), prefill ~420 pp. Binary: /app/llama-server-ternary (Q1_0_g128 kernels).
|
||||
"bonsai-27b-q1":
|
||||
name: "Bonsai-27B Q1_0 (1-bit ternary, all-GPU weights + KV in RAM)"
|
||||
description: "Small fast local model. 27B ternary @ 1-bit, 3.8GB weights fully on the 1650 Ti; KV offloaded to RAM (q4_0). ~11 t/s decode, ~420 pp prefill. Hybrid-thinking -> reasoning_content; give >=800 max_tokens."
|
||||
cmd: |
|
||||
${server-base} ${turbo2-kv}
|
||||
--model /models/ornith-9b-Q8_0.gguf
|
||||
--n-gpu-layers 11
|
||||
--ctx-size 32768
|
||||
--batch-size 512 --ubatch-size 256
|
||||
--cont-batching --parallel 1
|
||||
--no-mmap --mlock --jinja
|
||||
ttl: 300
|
||||
|
||||
"qwen35-9b":
|
||||
name: "Qwen3.5-9B Q8_0"
|
||||
description: "Reasoning distill. 4.38 t/s measured, mlock-pinned"
|
||||
cmd: |
|
||||
${server-base} ${turbo2-kv}
|
||||
--model /models/Qwen3.5-9B.Q8_0.gguf
|
||||
--n-gpu-layers 11
|
||||
--ctx-size 32768
|
||||
--batch-size 512 --ubatch-size 256
|
||||
--cont-batching --parallel 1
|
||||
--no-mmap --mlock
|
||||
ttl: 300
|
||||
|
||||
# ── Pure-GPU small models ───────────────────────────────────────────────────
|
||||
|
||||
"qwen3-4b":
|
||||
name: "Qwen3-4B Q4_K_M"
|
||||
description: "44 t/s @ 16K. NEVER turbo KV (PPL 438 @ 32K — FINDINGS.md §2)"
|
||||
cmd: |
|
||||
${server-base} ${q4-kv}
|
||||
--model /models/Qwen3-4B-Q4_K_M.gguf
|
||||
--n-gpu-layers 99
|
||||
/app/llama-server-ternary
|
||||
--host 127.0.0.1 --port ${PORT}
|
||||
--threads 6 --threads-batch 6
|
||||
--model /models/Bonsai-27B-Q1_0.gguf
|
||||
--n-gpu-layers 99 --main-gpu 0
|
||||
--flash-attn on
|
||||
--cache-type-k q4_0 --cache-type-v q4_0
|
||||
--no-kv-offload
|
||||
--batch-size 64 --ubatch-size 64
|
||||
--ctx-size 16384
|
||||
--batch-size 512 --ubatch-size 256
|
||||
--cont-batching --parallel 1
|
||||
ttl: 300
|
||||
|
||||
"smollm3-3b":
|
||||
name: "SmolLM3-3B Q4_K_M"
|
||||
description: "58 t/s @ 32K, thinking+tools, 2 slots"
|
||||
cmd: |
|
||||
${server-base} ${turbo2-kv}
|
||||
--model /models/HuggingFaceTB_SmolLM3-3B-Q4_K_M.gguf
|
||||
--n-gpu-layers 99
|
||||
--ctx-size 32768
|
||||
--batch-size 1024 --ubatch-size 512
|
||||
--cont-batching --parallel 2
|
||||
ttl: 300
|
||||
|
||||
"gemma4-e2b":
|
||||
name: "Gemma 4 E2B Q4_K_M"
|
||||
description: "66 t/s, multimodal, 131K ctx (MQA tiny KV; f16 KV — turbo2 worse)"
|
||||
cmd: |
|
||||
${server-base}
|
||||
--cache-type-k f16 --cache-type-v f16
|
||||
--model /models/google_gemma-4-E2B-it-Q4_K_M.gguf
|
||||
--n-gpu-layers 99
|
||||
--ctx-size 131072
|
||||
--batch-size 1024 --ubatch-size 512
|
||||
--cont-batching --parallel 2
|
||||
ttl: 300
|
||||
|
||||
"gemma4-e4b":
|
||||
name: "Gemma 4 E4B Q4_K_M"
|
||||
description: "32 t/s @ 24K, multimodal. ngl=42 needs free VRAM"
|
||||
cmd: |
|
||||
${server-base} ${turbo2-kv}
|
||||
--model /models/google_gemma-4-E4B-it-Q4_K_M.gguf
|
||||
--n-gpu-layers 42
|
||||
--ctx-size 24576
|
||||
--batch-size 1024 --ubatch-size 512
|
||||
--cont-batching --parallel 1
|
||||
ttl: 300
|
||||
|
||||
# ── bigctx variants (-nkvo: KV in RAM over PCIe, ~8 GB/s) ──────────────────
|
||||
|
||||
"smollm3-3b-bigctx":
|
||||
name: "SmolLM3-3B bigctx 65K"
|
||||
description: "~15 t/s @ 50% fill, KV in RAM"
|
||||
cmd: |
|
||||
${server-base} ${turbo2-kv}
|
||||
--model /models/HuggingFaceTB_SmolLM3-3B-Q4_K_M.gguf
|
||||
--n-gpu-layers 99
|
||||
--ctx-size 65536
|
||||
--batch-size 512 --ubatch-size 256
|
||||
--cont-batching --parallel 1
|
||||
--no-kv-offload
|
||||
ttl: 300
|
||||
|
||||
"gemma4-e2b-bigctx":
|
||||
name: "Gemma 4 E2B bigctx 393K"
|
||||
description: "~17 t/s @ 50% fill. q4_0 KV (turbo2 worse on MQA)"
|
||||
cmd: |
|
||||
${server-base} ${q4-kv}
|
||||
--model /models/google_gemma-4-E2B-it-Q4_K_M.gguf
|
||||
--n-gpu-layers 99
|
||||
--ctx-size 393216
|
||||
--batch-size 512 --ubatch-size 256
|
||||
--cont-batching --parallel 1
|
||||
--no-kv-offload
|
||||
ttl: 300
|
||||
|
||||
"gemma4-e4b-bigctx":
|
||||
name: "Gemma 4 E4B bigctx 163K"
|
||||
description: "~18 t/s @ 50% fill, KV in RAM"
|
||||
cmd: |
|
||||
${server-base} ${turbo2-kv}
|
||||
--model /models/google_gemma-4-E4B-it-Q4_K_M.gguf
|
||||
--n-gpu-layers 42
|
||||
--ctx-size 163840
|
||||
--batch-size 512 --ubatch-size 128
|
||||
--cont-batching --parallel 1
|
||||
--no-kv-offload
|
||||
ttl: 300
|
||||
|
||||
"qwen3-4b-bigctx":
|
||||
name: "Qwen3-4B bigctx 24K"
|
||||
description: "~11 t/s @ 50% fill. q4_0 KV only (turbo broken)"
|
||||
cmd: |
|
||||
${server-base} ${q4-kv}
|
||||
--model /models/Qwen3-4B-Q4_K_M.gguf
|
||||
--n-gpu-layers 20
|
||||
--ctx-size 24576
|
||||
--batch-size 512 --ubatch-size 256
|
||||
--cont-batching --parallel 1
|
||||
--no-kv-offload
|
||||
--jinja
|
||||
ttl: 300
|
||||
|
||||
Reference in New Issue
Block a user