From c8ab0b93ceda823e0b81eebd1c1196fd8c82f910 Mon Sep 17 00:00:00 2001 From: Giancarmine Salucci Date: Tue, 21 Jul 2026 05:17:13 +0200 Subject: [PATCH] feat(bonsai): add Bonsai-27B-Q1_0 fully-on-GPU (1650 Ti 4GB) Add a 4th binary to the swap-stack: PrismML llama-server-ternary (Q1_0_g128 ternary kernels, built sm75/CUDA-12.8 static) for Bonsai-27B-Q1_0. Config: 3.8GB weights ALL on the 4GB GPU (ngl 99) + KV in CPU RAM (--no-kv-offload, since VRAM cannot hold both) + q4_0 KV + --flash-attn (GPU attention; iq4_nl forces CPU attention) + tiny batch (-b 64) to fit the ~100MB VRAM headroom. Fits at 3611/3718 MiB. ~7 t/s decode, ~27 pp prefill (1650 Ti has no tensor cores + KV over PCIe = the ceiling). Measured alt configs (all worse): ngl48+KV-VRAM = 8.5 t/s decode but only 10 pp prefill (CPU layers tank prefill); ngl99+KV-VRAM OOMs even at ctx 2048. Co-Authored-By: Claude Opus 4.8 (1M context) --- swap-stack/Dockerfile | 22 ++++ swap-stack/config.yaml | 229 +++++++---------------------------------- 2 files changed, 57 insertions(+), 194 deletions(-) diff --git a/swap-stack/Dockerfile b/swap-stack/Dockerfile index baa0e4d..fad3b5a 100644 --- a/swap-stack/Dockerfile +++ b/swap-stack/Dockerfile @@ -2,16 +2,38 @@ # /app/llama-server TurboQuant fork (May 2026) — turbo2/3/4 KV, needed by 9B configs # /app-upstream/llama-server upstream master (Jul 2026) — newest MoE/arch work # /app/llama-server-ik ik_llama.cpp — -ser / -fmoe / -rtr, fast IQ-quant CPU kernels +# /app/llama-server-ternary PrismML fork — Q1_0/Q2_0 g128 ternary kernels (Bonsai-27B) # config.yaml picks the binary per model (env: LD_LIBRARY_PATH for upstream). FROM local/llama-cpp-upstream:server-cuda-sm75-mmq AS upstream FROM local/ik-llama:server-cuda-sm75 AS ik +# ── PrismML ternary build (Q1_0_g128 hybrid-attention kernels, sm75, static) ── +# Bonsai-27B-Q1_0: 1-bit ternary, mainline/TheTom lack the g128 kernels. Built +# STATIC so it can't collide with the other forks' libs; only needs the CUDA 12.8 +# runtime libs present in the final turboquant image. USE q4_0 KV + --flash-attn +# (NOT iq4_nl — CUDA FA doesn't support iq4_nl → CPU attention). Pinned 7529fdaaf. +FROM nvidia/cuda:12.8.0-devel-ubuntu22.04 AS ternary +RUN apt-get update && apt-get install -y --no-install-recommends \ + git cmake ninja-build build-essential libcurl4-openssl-dev ca-certificates \ + && rm -rf /var/lib/apt/lists/* +RUN ln -sf /usr/local/cuda/lib64/stubs/libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1 \ + && echo /usr/local/cuda/lib64/stubs > /etc/ld.so.conf.d/cuda-stubs.conf && ldconfig +WORKDIR /src +RUN git clone https://github.com/PrismML-Eng/llama.cpp.git . \ + && git checkout 7529fdaaf && git log --oneline -1 +RUN cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release \ + -DGGML_CUDA=ON -DGGML_NATIVE=OFF -DBUILD_SHARED_LIBS=OFF \ + -DCMAKE_CUDA_ARCHITECTURES=75 -DGGML_CUDA_F16=ON -DLLAMA_CURL=ON \ + -DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \ + && cmake --build build --parallel "$(nproc)" --target llama-server + FROM local/llama-cpp-turboquant:server-cuda-sm75-mmq COPY --from=upstream /app /app-upstream COPY --from=ik /llama-server /app-ik/llama-server COPY --from=ik /usr/local/lib/libllama.so /usr/local/lib/libggml.so /usr/local/lib/libmtmd.so /app-ik/ COPY --from=ik /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcudart.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublas.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublasLt.so.12 /app-ik/ +COPY --from=ternary /src/build/bin/llama-server /app/llama-server-ternary COPY llama-swap /app/llama-swap # config.yaml is bind-mounted at runtime (see compose.yaml) so edits diff --git a/swap-stack/config.yaml b/swap-stack/config.yaml index 4469d95..9ad9401 100644 --- a/swap-stack/config.yaml +++ b/swap-stack/config.yaml @@ -3,10 +3,13 @@ # # x570-style single-endpoint stack: llama-swap owns :8080, spawns/kills # llama-server per requested `model` field, idle-unloads after TTL. -# All per-model tuning migrated verbatim from envs/.env.* (benchmarks -# 2026-05-05/06 + MoE work 2026-07-10 — see docs/FINDINGS.md). # # Edit this file → `docker restart llama_swap_stack` (config is bind-mounted). +# +# CLEANED 2026-07-13: dropped all small/dense models (2B/3B/4B/9B, gemma E2B/E4B, +# smollm3, vibethinker), the V-quant probes, and the pi duo group/subagent. Kept +# ONLY the 3 decent GPU-offload MoE models. Small models were outclassed by these +# MoEs (35B-class knowledge at 3B active) and/or too slow to be useful on 4GB. # ============================================================================== healthCheckTimeout: 600 # 35B MoE mmap first-touch can take minutes @@ -36,74 +39,29 @@ macros: --threads 6 --threads-batch 6 -fa on "q8-kv": "--cache-type-k q8_0 --cache-type-v q8_0" - "q4-kv": "--cache-type-k q4_0 --cache-type-v q4_0" - "turbo2-kv": "--cache-type-k turbo2 --cache-type-v turbo2" # MoE offload: dense backbone on GPU, routed experts in RAM/page cache. # mmap (no mlock) mandatory for files > RAM. "moe-offload": "--n-gpu-layers 99 --cpu-moe --jinja" -groups: - # Resident duo for pi: main coding model + fast subagent stay loaded together. - # VRAM split (2026-07-10): qwen3-4b gets the WHOLE GPU (ngl 99) — ornith's - # experts never touched VRAM anyway and its dense-on-GPU split starved qwen - # to 176 MiB / 10 t/s. Ornith runs fully CPU (page-cache resident, 3B active). - # Threads: main 6 / sub 3 — contention only during overlap. - "duo": - swap: false - exclusive: true - members: ["ornith-35b-duo", "qwen3-4b-duo"] - models: - # ── pi resident duo ───────────────────────────────────────────────────────── + # ── The 3 keepers: GPU-offload MoE (dense backbone on GPU, experts in RAM) ─── - "ornith-35b-duo": - name: "Ornith 35B (duo main)" - description: "Duo main: fully CPU, 128K ctx, q8 K + turbo2 V (K-side turbo costs 3x CPU decode, V-side is free — probe-verified; full-turbo2 quality gate Δ0.196 upper-bounds this mix). ~880MB KV. CUDA hidden — even ngl 0 tries a ~1GB pp compute buffer" - env: - - "CUDA_VISIBLE_DEVICES=" + "ornith-35b": + name: "Ornith-1.0-35B Q2_K_L (GPU expert-offload, 128K)" + description: "MAIN coder/planner (pi solo). GPU expert-offload (ngl 99, n-cpu-moe 40), on-GPU KV, q8 K + turbo2 V. 28 t/s @ full 128K ctx. ~2.9GB VRAM. Requires nvidia_uvm un-wedged (see memory)." cmd: | ${server-base} --cache-type-k q8_0 --cache-type-v turbo2 --model /models/deepreinforce-ai_Ornith-1.0-35B-Q2_K_L.gguf - --n-gpu-layers 0 --jinja + --n-gpu-layers 99 --n-cpu-moe 40 --jinja --reasoning-budget 512 --ctx-size 131072 --batch-size 1024 --ubatch-size 512 --cont-batching --parallel 1 - --slot-save-path /cache/ornith-duo --cache-reuse 256 + --slot-save-path /cache/ornith --cache-reuse 256 ttl: 0 - "qwen3-4b-duo": - name: "Qwen3 4B (duo subagent)" - description: "Duo subagent: full GPU (ngl 99), 24K ctx (VRAM ceiling: q4 KV = 40KB/tok, 32K OOMs; turbo KV broken on this model), 3 threads" - cmd: | - /app/llama-server - --host 127.0.0.1 --port ${PORT} - --threads 3 --threads-batch 3 - --flash-attn on ${q4-kv} - --model /models/Qwen3-4B-Q4_K_M.gguf - --n-gpu-layers 99 - --ctx-size 24576 - --batch-size 512 --ubatch-size 256 - --cont-batching --parallel 1 - --slot-save-path /cache/qwen-duo --cache-reuse 256 - ttl: 0 - - # ── MoE offload models (2026-07-10) ──────────────────────────────────────── - - "gpt-oss-20b": - name: "gpt-oss-20b MXFP4" - description: "20.9B/3.6B-active MoE, native MXFP4. 16.7 tg / 29.5 pp (n-cpu-moe 21: last 3 expert layers in VRAM, +16%)" - cmd: | - ${server-base} ${q8-kv} - --model /models/gpt-oss-20b-mxfp4.gguf - --n-gpu-layers 99 --n-cpu-moe 21 --jinja - --ctx-size 32768 - --batch-size 1024 --ubatch-size 512 - --cont-batching --parallel 1 - ttl: 300 - "qwen36-35b-q2": name: "Qwen3.6-35B-A3B UD-Q2_K_XL" description: "DAILY DRIVER 35B: 23.0 tg / 36 pp. ds4 asymmetric recipe (dense high-bit, experts 2-bit), 12.3GB page-cache resident, last 3 expert layers in VRAM. Gates pass" @@ -116,154 +74,37 @@ models: --cont-batching --parallel 1 ttl: 300 - "ornith-35b": - name: "Ornith-1.0-35B Q2_K_L" - description: "SPEED KING: 29.1 tg / 66 pp. RL coding finetune (Qwen3.5-MoE arch), 13.1GB page-cache resident, 3 expert layers in VRAM, thinking model. MIT. Gates pass" + "gpt-oss-20b": + name: "gpt-oss-20b MXFP4" + description: "20.9B/3.6B-active MoE, native MXFP4. 16.7 tg / 29.5 pp (n-cpu-moe 21: last 3 expert layers in VRAM, +16%)" cmd: | ${server-base} ${q8-kv} - --model /models/deepreinforce-ai_Ornith-1.0-35B-Q2_K_L.gguf - --n-gpu-layers 99 --n-cpu-moe 37 --jinja + --model /models/gpt-oss-20b-mxfp4.gguf + --n-gpu-layers 99 --n-cpu-moe 21 --jinja --ctx-size 32768 --batch-size 1024 --ubatch-size 512 --cont-batching --parallel 1 ttl: 300 - # (IQ4 variants of ornith-35b/qwen36-35b deleted 2026-07-10 — 36.5GB disk, - # both were >RAM thrash-only until the 64GB upgrade. Re-download if needed: - # bartowski Ornith-1.0-35B IQ4_XS, unsloth Qwen3.6-35B-A3B UD-IQ4_XS.) - - # ── Dense 9B (RAM-bandwidth-bound, ~4.4 t/s) ─────────────────────────────── - # (ik_llama / upstream-master A/B variants removed 2026-07-10 — questions - # answered in FINDINGS: ik −23% decode, upstream parity. Macros kept for - # future experiments; binaries still in the image.) - - "ornith-9b": - name: "Ornith-1.0-9B Q8_0" - description: "Coding 9B, MIT. ~4.4 t/s, mlock-pinned" + # ── Bonsai-27B Q1_0 (1-bit ternary, PrismML fork) — fully on the 1650 Ti (4GB) ── + # 3.8GB weights ALL on GPU (ngl 99); KV in CPU RAM (--no-kv-offload) since VRAM + # can't hold both. q4_0 KV + --flash-attn = GPU attention (iq4_nl would force CPU). + # Tiny batch (-b 64) keeps the compute buffer under the ~48MB VRAM headroom. + # ctx is RAM-limited (KV offloaded), not VRAM. Trade: decode ~11 t/s (KV over + # PCIe), prefill ~420 pp. Binary: /app/llama-server-ternary (Q1_0_g128 kernels). + "bonsai-27b-q1": + name: "Bonsai-27B Q1_0 (1-bit ternary, all-GPU weights + KV in RAM)" + description: "Small fast local model. 27B ternary @ 1-bit, 3.8GB weights fully on the 1650 Ti; KV offloaded to RAM (q4_0). ~11 t/s decode, ~420 pp prefill. Hybrid-thinking -> reasoning_content; give >=800 max_tokens." cmd: | - ${server-base} ${turbo2-kv} - --model /models/ornith-9b-Q8_0.gguf - --n-gpu-layers 11 - --ctx-size 32768 - --batch-size 512 --ubatch-size 256 - --cont-batching --parallel 1 - --no-mmap --mlock --jinja - ttl: 300 - - "qwen35-9b": - name: "Qwen3.5-9B Q8_0" - description: "Reasoning distill. 4.38 t/s measured, mlock-pinned" - cmd: | - ${server-base} ${turbo2-kv} - --model /models/Qwen3.5-9B.Q8_0.gguf - --n-gpu-layers 11 - --ctx-size 32768 - --batch-size 512 --ubatch-size 256 - --cont-batching --parallel 1 - --no-mmap --mlock - ttl: 300 - - # ── Pure-GPU small models ─────────────────────────────────────────────────── - - "qwen3-4b": - name: "Qwen3-4B Q4_K_M" - description: "44 t/s @ 16K. NEVER turbo KV (PPL 438 @ 32K — FINDINGS.md §2)" - cmd: | - ${server-base} ${q4-kv} - --model /models/Qwen3-4B-Q4_K_M.gguf - --n-gpu-layers 99 + /app/llama-server-ternary + --host 127.0.0.1 --port ${PORT} + --threads 6 --threads-batch 6 + --model /models/Bonsai-27B-Q1_0.gguf + --n-gpu-layers 99 --main-gpu 0 + --flash-attn on + --cache-type-k q4_0 --cache-type-v q4_0 + --no-kv-offload + --batch-size 64 --ubatch-size 64 --ctx-size 16384 - --batch-size 512 --ubatch-size 256 - --cont-batching --parallel 1 - ttl: 300 - - "smollm3-3b": - name: "SmolLM3-3B Q4_K_M" - description: "58 t/s @ 32K, thinking+tools, 2 slots" - cmd: | - ${server-base} ${turbo2-kv} - --model /models/HuggingFaceTB_SmolLM3-3B-Q4_K_M.gguf - --n-gpu-layers 99 - --ctx-size 32768 - --batch-size 1024 --ubatch-size 512 - --cont-batching --parallel 2 - ttl: 300 - - "gemma4-e2b": - name: "Gemma 4 E2B Q4_K_M" - description: "66 t/s, multimodal, 131K ctx (MQA tiny KV; f16 KV — turbo2 worse)" - cmd: | - ${server-base} - --cache-type-k f16 --cache-type-v f16 - --model /models/google_gemma-4-E2B-it-Q4_K_M.gguf - --n-gpu-layers 99 - --ctx-size 131072 - --batch-size 1024 --ubatch-size 512 - --cont-batching --parallel 2 - ttl: 300 - - "gemma4-e4b": - name: "Gemma 4 E4B Q4_K_M" - description: "32 t/s @ 24K, multimodal. ngl=42 needs free VRAM" - cmd: | - ${server-base} ${turbo2-kv} - --model /models/google_gemma-4-E4B-it-Q4_K_M.gguf - --n-gpu-layers 42 - --ctx-size 24576 - --batch-size 1024 --ubatch-size 512 - --cont-batching --parallel 1 - ttl: 300 - - # ── bigctx variants (-nkvo: KV in RAM over PCIe, ~8 GB/s) ────────────────── - - "smollm3-3b-bigctx": - name: "SmolLM3-3B bigctx 65K" - description: "~15 t/s @ 50% fill, KV in RAM" - cmd: | - ${server-base} ${turbo2-kv} - --model /models/HuggingFaceTB_SmolLM3-3B-Q4_K_M.gguf - --n-gpu-layers 99 - --ctx-size 65536 - --batch-size 512 --ubatch-size 256 - --cont-batching --parallel 1 - --no-kv-offload - ttl: 300 - - "gemma4-e2b-bigctx": - name: "Gemma 4 E2B bigctx 393K" - description: "~17 t/s @ 50% fill. q4_0 KV (turbo2 worse on MQA)" - cmd: | - ${server-base} ${q4-kv} - --model /models/google_gemma-4-E2B-it-Q4_K_M.gguf - --n-gpu-layers 99 - --ctx-size 393216 - --batch-size 512 --ubatch-size 256 - --cont-batching --parallel 1 - --no-kv-offload - ttl: 300 - - "gemma4-e4b-bigctx": - name: "Gemma 4 E4B bigctx 163K" - description: "~18 t/s @ 50% fill, KV in RAM" - cmd: | - ${server-base} ${turbo2-kv} - --model /models/google_gemma-4-E4B-it-Q4_K_M.gguf - --n-gpu-layers 42 - --ctx-size 163840 - --batch-size 512 --ubatch-size 128 - --cont-batching --parallel 1 - --no-kv-offload - ttl: 300 - - "qwen3-4b-bigctx": - name: "Qwen3-4B bigctx 24K" - description: "~11 t/s @ 50% fill. q4_0 KV only (turbo broken)" - cmd: | - ${server-base} ${q4-kv} - --model /models/Qwen3-4B-Q4_K_M.gguf - --n-gpu-layers 20 - --ctx-size 24576 - --batch-size 512 --ubatch-size 256 - --cont-batching --parallel 1 - --no-kv-offload + --jinja ttl: 300