feat(bonsai): add Bonsai-27B-Q1_0 fully-on-GPU (1650 Ti 4GB)

Add a 4th binary to the swap-stack: PrismML llama-server-ternary (Q1_0_g128
ternary kernels, built sm75/CUDA-12.8 static) for Bonsai-27B-Q1_0.

Config: 3.8GB weights ALL on the 4GB GPU (ngl 99) + KV in CPU RAM
(--no-kv-offload, since VRAM cannot hold both) + q4_0 KV + --flash-attn
(GPU attention; iq4_nl forces CPU attention) + tiny batch (-b 64) to fit
the ~100MB VRAM headroom. Fits at 3611/3718 MiB. ~7 t/s decode, ~27 pp
prefill (1650 Ti has no tensor cores + KV over PCIe = the ceiling).

Measured alt configs (all worse): ngl48+KV-VRAM = 8.5 t/s decode but only
10 pp prefill (CPU layers tank prefill); ngl99+KV-VRAM OOMs even at ctx 2048.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-07-21 05:17:13 +02:00
co-authored by Claude Opus 4.8
parent c7bb54a253
commit c8ab0b93ce
2 changed files with 57 additions and 194 deletions
+22
View File
@@ -2,16 +2,38 @@
# /app/llama-server TurboQuant fork (May 2026) — turbo2/3/4 KV, needed by 9B configs
# /app-upstream/llama-server upstream master (Jul 2026) — newest MoE/arch work
# /app/llama-server-ik ik_llama.cpp — -ser / -fmoe / -rtr, fast IQ-quant CPU kernels
# /app/llama-server-ternary PrismML fork — Q1_0/Q2_0 g128 ternary kernels (Bonsai-27B)
# config.yaml picks the binary per model (env: LD_LIBRARY_PATH for upstream).
FROM local/llama-cpp-upstream:server-cuda-sm75-mmq AS upstream
FROM local/ik-llama:server-cuda-sm75 AS ik
# ── PrismML ternary build (Q1_0_g128 hybrid-attention kernels, sm75, static) ──
# Bonsai-27B-Q1_0: 1-bit ternary, mainline/TheTom lack the g128 kernels. Built
# STATIC so it can't collide with the other forks' libs; only needs the CUDA 12.8
# runtime libs present in the final turboquant image. USE q4_0 KV + --flash-attn
# (NOT iq4_nl — CUDA FA doesn't support iq4_nl → CPU attention). Pinned 7529fdaaf.
FROM nvidia/cuda:12.8.0-devel-ubuntu22.04 AS ternary
RUN apt-get update && apt-get install -y --no-install-recommends \
git cmake ninja-build build-essential libcurl4-openssl-dev ca-certificates \
&& rm -rf /var/lib/apt/lists/*
RUN ln -sf /usr/local/cuda/lib64/stubs/libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1 \
&& echo /usr/local/cuda/lib64/stubs > /etc/ld.so.conf.d/cuda-stubs.conf && ldconfig
WORKDIR /src
RUN git clone https://github.com/PrismML-Eng/llama.cpp.git . \
&& git checkout 7529fdaaf && git log --oneline -1
RUN cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release \
-DGGML_CUDA=ON -DGGML_NATIVE=OFF -DBUILD_SHARED_LIBS=OFF \
-DCMAKE_CUDA_ARCHITECTURES=75 -DGGML_CUDA_F16=ON -DLLAMA_CURL=ON \
-DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \
&& cmake --build build --parallel "$(nproc)" --target llama-server
FROM local/llama-cpp-turboquant:server-cuda-sm75-mmq
COPY --from=upstream /app /app-upstream
COPY --from=ik /llama-server /app-ik/llama-server
COPY --from=ik /usr/local/lib/libllama.so /usr/local/lib/libggml.so /usr/local/lib/libmtmd.so /app-ik/
COPY --from=ik /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcudart.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublas.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublasLt.so.12 /app-ik/
COPY --from=ternary /src/build/bin/llama-server /app/llama-server-ternary
COPY llama-swap /app/llama-swap
# config.yaml is bind-mounted at runtime (see compose.yaml) so edits
+35 -194
View File
@@ -3,10 +3,13 @@
#
# x570-style single-endpoint stack: llama-swap owns :8080, spawns/kills
# llama-server per requested `model` field, idle-unloads after TTL.
# All per-model tuning migrated verbatim from envs/.env.* (benchmarks
# 2026-05-05/06 + MoE work 2026-07-10 — see docs/FINDINGS.md).
#
# Edit this file → `docker restart llama_swap_stack` (config is bind-mounted).
#
# CLEANED 2026-07-13: dropped all small/dense models (2B/3B/4B/9B, gemma E2B/E4B,
# smollm3, vibethinker), the V-quant probes, and the pi duo group/subagent. Kept
# ONLY the 3 decent GPU-offload MoE models. Small models were outclassed by these
# MoEs (35B-class knowledge at 3B active) and/or too slow to be useful on 4GB.
# ==============================================================================
healthCheckTimeout: 600 # 35B MoE mmap first-touch can take minutes
@@ -36,74 +39,29 @@ macros:
--threads 6 --threads-batch 6 -fa on
"q8-kv": "--cache-type-k q8_0 --cache-type-v q8_0"
"q4-kv": "--cache-type-k q4_0 --cache-type-v q4_0"
"turbo2-kv": "--cache-type-k turbo2 --cache-type-v turbo2"
# MoE offload: dense backbone on GPU, routed experts in RAM/page cache.
# mmap (no mlock) mandatory for files > RAM.
"moe-offload": "--n-gpu-layers 99 --cpu-moe --jinja"
groups:
# Resident duo for pi: main coding model + fast subagent stay loaded together.
# VRAM split (2026-07-10): qwen3-4b gets the WHOLE GPU (ngl 99) — ornith's
# experts never touched VRAM anyway and its dense-on-GPU split starved qwen
# to 176 MiB / 10 t/s. Ornith runs fully CPU (page-cache resident, 3B active).
# Threads: main 6 / sub 3 — contention only during overlap.
"duo":
swap: false
exclusive: true
members: ["ornith-35b-duo", "qwen3-4b-duo"]
models:
# ── pi resident duo ─────────────────────────────────────────────────────────
# ── The 3 keepers: GPU-offload MoE (dense backbone on GPU, experts in RAM) ───
"ornith-35b-duo":
name: "Ornith 35B (duo main)"
description: "Duo main: fully CPU, 128K ctx, q8 K + turbo2 V (K-side turbo costs 3x CPU decode, V-side is free — probe-verified; full-turbo2 quality gate Δ0.196 upper-bounds this mix). ~880MB KV. CUDA hidden — even ngl 0 tries a ~1GB pp compute buffer"
env:
- "CUDA_VISIBLE_DEVICES="
"ornith-35b":
name: "Ornith-1.0-35B Q2_K_L (GPU expert-offload, 128K)"
description: "MAIN coder/planner (pi solo). GPU expert-offload (ngl 99, n-cpu-moe 40), on-GPU KV, q8 K + turbo2 V. 28 t/s @ full 128K ctx. ~2.9GB VRAM. Requires nvidia_uvm un-wedged (see memory)."
cmd: |
${server-base}
--cache-type-k q8_0 --cache-type-v turbo2
--model /models/deepreinforce-ai_Ornith-1.0-35B-Q2_K_L.gguf
--n-gpu-layers 0 --jinja
--n-gpu-layers 99 --n-cpu-moe 40 --jinja --reasoning-budget 512
--ctx-size 131072
--batch-size 1024 --ubatch-size 512
--cont-batching --parallel 1
--slot-save-path /cache/ornith-duo --cache-reuse 256
--slot-save-path /cache/ornith --cache-reuse 256
ttl: 0
"qwen3-4b-duo":
name: "Qwen3 4B (duo subagent)"
description: "Duo subagent: full GPU (ngl 99), 24K ctx (VRAM ceiling: q4 KV = 40KB/tok, 32K OOMs; turbo KV broken on this model), 3 threads"
cmd: |
/app/llama-server
--host 127.0.0.1 --port ${PORT}
--threads 3 --threads-batch 3
--flash-attn on ${q4-kv}
--model /models/Qwen3-4B-Q4_K_M.gguf
--n-gpu-layers 99
--ctx-size 24576
--batch-size 512 --ubatch-size 256
--cont-batching --parallel 1
--slot-save-path /cache/qwen-duo --cache-reuse 256
ttl: 0
# ── MoE offload models (2026-07-10) ────────────────────────────────────────
"gpt-oss-20b":
name: "gpt-oss-20b MXFP4"
description: "20.9B/3.6B-active MoE, native MXFP4. 16.7 tg / 29.5 pp (n-cpu-moe 21: last 3 expert layers in VRAM, +16%)"
cmd: |
${server-base} ${q8-kv}
--model /models/gpt-oss-20b-mxfp4.gguf
--n-gpu-layers 99 --n-cpu-moe 21 --jinja
--ctx-size 32768
--batch-size 1024 --ubatch-size 512
--cont-batching --parallel 1
ttl: 300
"qwen36-35b-q2":
name: "Qwen3.6-35B-A3B UD-Q2_K_XL"
description: "DAILY DRIVER 35B: 23.0 tg / 36 pp. ds4 asymmetric recipe (dense high-bit, experts 2-bit), 12.3GB page-cache resident, last 3 expert layers in VRAM. Gates pass"
@@ -116,154 +74,37 @@ models:
--cont-batching --parallel 1
ttl: 300
"ornith-35b":
name: "Ornith-1.0-35B Q2_K_L"
description: "SPEED KING: 29.1 tg / 66 pp. RL coding finetune (Qwen3.5-MoE arch), 13.1GB page-cache resident, 3 expert layers in VRAM, thinking model. MIT. Gates pass"
"gpt-oss-20b":
name: "gpt-oss-20b MXFP4"
description: "20.9B/3.6B-active MoE, native MXFP4. 16.7 tg / 29.5 pp (n-cpu-moe 21: last 3 expert layers in VRAM, +16%)"
cmd: |
${server-base} ${q8-kv}
--model /models/deepreinforce-ai_Ornith-1.0-35B-Q2_K_L.gguf
--n-gpu-layers 99 --n-cpu-moe 37 --jinja
--model /models/gpt-oss-20b-mxfp4.gguf
--n-gpu-layers 99 --n-cpu-moe 21 --jinja
--ctx-size 32768
--batch-size 1024 --ubatch-size 512
--cont-batching --parallel 1
ttl: 300
# (IQ4 variants of ornith-35b/qwen36-35b deleted 2026-07-10 — 36.5GB disk,
# both were >RAM thrash-only until the 64GB upgrade. Re-download if needed:
# bartowski Ornith-1.0-35B IQ4_XS, unsloth Qwen3.6-35B-A3B UD-IQ4_XS.)
# ── Dense 9B (RAM-bandwidth-bound, ~4.4 t/s) ───────────────────────────────
# (ik_llama / upstream-master A/B variants removed 2026-07-10 — questions
# answered in FINDINGS: ik −23% decode, upstream parity. Macros kept for
# future experiments; binaries still in the image.)
"ornith-9b":
name: "Ornith-1.0-9B Q8_0"
description: "Coding 9B, MIT. ~4.4 t/s, mlock-pinned"
# ── Bonsai-27B Q1_0 (1-bit ternary, PrismML fork) — fully on the 1650 Ti (4GB) ──
# 3.8GB weights ALL on GPU (ngl 99); KV in CPU RAM (--no-kv-offload) since VRAM
# can't hold both. q4_0 KV + --flash-attn = GPU attention (iq4_nl would force CPU).
# Tiny batch (-b 64) keeps the compute buffer under the ~48MB VRAM headroom.
# ctx is RAM-limited (KV offloaded), not VRAM. Trade: decode ~11 t/s (KV over
# PCIe), prefill ~420 pp. Binary: /app/llama-server-ternary (Q1_0_g128 kernels).
"bonsai-27b-q1":
name: "Bonsai-27B Q1_0 (1-bit ternary, all-GPU weights + KV in RAM)"
description: "Small fast local model. 27B ternary @ 1-bit, 3.8GB weights fully on the 1650 Ti; KV offloaded to RAM (q4_0). ~11 t/s decode, ~420 pp prefill. Hybrid-thinking -> reasoning_content; give >=800 max_tokens."
cmd: |
${server-base} ${turbo2-kv}
--model /models/ornith-9b-Q8_0.gguf
--n-gpu-layers 11
--ctx-size 32768
--batch-size 512 --ubatch-size 256
--cont-batching --parallel 1
--no-mmap --mlock --jinja
ttl: 300
"qwen35-9b":
name: "Qwen3.5-9B Q8_0"
description: "Reasoning distill. 4.38 t/s measured, mlock-pinned"
cmd: |
${server-base} ${turbo2-kv}
--model /models/Qwen3.5-9B.Q8_0.gguf
--n-gpu-layers 11
--ctx-size 32768
--batch-size 512 --ubatch-size 256
--cont-batching --parallel 1
--no-mmap --mlock
ttl: 300
# ── Pure-GPU small models ───────────────────────────────────────────────────
"qwen3-4b":
name: "Qwen3-4B Q4_K_M"
description: "44 t/s @ 16K. NEVER turbo KV (PPL 438 @ 32K — FINDINGS.md §2)"
cmd: |
${server-base} ${q4-kv}
--model /models/Qwen3-4B-Q4_K_M.gguf
--n-gpu-layers 99
/app/llama-server-ternary
--host 127.0.0.1 --port ${PORT}
--threads 6 --threads-batch 6
--model /models/Bonsai-27B-Q1_0.gguf
--n-gpu-layers 99 --main-gpu 0
--flash-attn on
--cache-type-k q4_0 --cache-type-v q4_0
--no-kv-offload
--batch-size 64 --ubatch-size 64
--ctx-size 16384
--batch-size 512 --ubatch-size 256
--cont-batching --parallel 1
ttl: 300
"smollm3-3b":
name: "SmolLM3-3B Q4_K_M"
description: "58 t/s @ 32K, thinking+tools, 2 slots"
cmd: |
${server-base} ${turbo2-kv}
--model /models/HuggingFaceTB_SmolLM3-3B-Q4_K_M.gguf
--n-gpu-layers 99
--ctx-size 32768
--batch-size 1024 --ubatch-size 512
--cont-batching --parallel 2
ttl: 300
"gemma4-e2b":
name: "Gemma 4 E2B Q4_K_M"
description: "66 t/s, multimodal, 131K ctx (MQA tiny KV; f16 KV — turbo2 worse)"
cmd: |
${server-base}
--cache-type-k f16 --cache-type-v f16
--model /models/google_gemma-4-E2B-it-Q4_K_M.gguf
--n-gpu-layers 99
--ctx-size 131072
--batch-size 1024 --ubatch-size 512
--cont-batching --parallel 2
ttl: 300
"gemma4-e4b":
name: "Gemma 4 E4B Q4_K_M"
description: "32 t/s @ 24K, multimodal. ngl=42 needs free VRAM"
cmd: |
${server-base} ${turbo2-kv}
--model /models/google_gemma-4-E4B-it-Q4_K_M.gguf
--n-gpu-layers 42
--ctx-size 24576
--batch-size 1024 --ubatch-size 512
--cont-batching --parallel 1
ttl: 300
# ── bigctx variants (-nkvo: KV in RAM over PCIe, ~8 GB/s) ──────────────────
"smollm3-3b-bigctx":
name: "SmolLM3-3B bigctx 65K"
description: "~15 t/s @ 50% fill, KV in RAM"
cmd: |
${server-base} ${turbo2-kv}
--model /models/HuggingFaceTB_SmolLM3-3B-Q4_K_M.gguf
--n-gpu-layers 99
--ctx-size 65536
--batch-size 512 --ubatch-size 256
--cont-batching --parallel 1
--no-kv-offload
ttl: 300
"gemma4-e2b-bigctx":
name: "Gemma 4 E2B bigctx 393K"
description: "~17 t/s @ 50% fill. q4_0 KV (turbo2 worse on MQA)"
cmd: |
${server-base} ${q4-kv}
--model /models/google_gemma-4-E2B-it-Q4_K_M.gguf
--n-gpu-layers 99
--ctx-size 393216
--batch-size 512 --ubatch-size 256
--cont-batching --parallel 1
--no-kv-offload
ttl: 300
"gemma4-e4b-bigctx":
name: "Gemma 4 E4B bigctx 163K"
description: "~18 t/s @ 50% fill, KV in RAM"
cmd: |
${server-base} ${turbo2-kv}
--model /models/google_gemma-4-E4B-it-Q4_K_M.gguf
--n-gpu-layers 42
--ctx-size 163840
--batch-size 512 --ubatch-size 128
--cont-batching --parallel 1
--no-kv-offload
ttl: 300
"qwen3-4b-bigctx":
name: "Qwen3-4B bigctx 24K"
description: "~11 t/s @ 50% fill. q4_0 KV only (turbo broken)"
cmd: |
${server-base} ${q4-kv}
--model /models/Qwen3-4B-Q4_K_M.gguf
--n-gpu-layers 20
--ctx-size 24576
--batch-size 512 --ubatch-size 256
--cont-batching --parallel 1
--no-kv-offload
--jinja
ttl: 300