Add a 4th binary to the swap-stack: PrismML llama-server-ternary (Q1_0_g128 ternary kernels, built sm75/CUDA-12.8 static) for Bonsai-27B-Q1_0. Config: 3.8GB weights ALL on the 4GB GPU (ngl 99) + KV in CPU RAM (--no-kv-offload, since VRAM cannot hold both) + q4_0 KV + --flash-attn (GPU attention; iq4_nl forces CPU attention) + tiny batch (-b 64) to fit the ~100MB VRAM headroom. Fits at 3611/3718 MiB. ~7 t/s decode, ~27 pp prefill (1650 Ti has no tensor cores + KV over PCIe = the ceiling). Measured alt configs (all worse): ngl48+KV-VRAM = 8.5 t/s decode but only 10 pp prefill (CPU layers tank prefill); ngl99+KV-VRAM OOMs even at ctx 2048. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
111 lines
4.6 KiB
YAML
111 lines
4.6 KiB
YAML
# ==============================================================================
|
|
# llama-swap config — xps9700 (GTX 1650 Ti 3.7GB VRAM, i7-10750H 6c/12t, 15 GiB)
|
|
#
|
|
# x570-style single-endpoint stack: llama-swap owns :8080, spawns/kills
|
|
# llama-server per requested `model` field, idle-unloads after TTL.
|
|
#
|
|
# Edit this file → `docker restart llama_swap_stack` (config is bind-mounted).
|
|
#
|
|
# CLEANED 2026-07-13: dropped all small/dense models (2B/3B/4B/9B, gemma E2B/E4B,
|
|
# smollm3, vibethinker), the V-quant probes, and the pi duo group/subagent. Kept
|
|
# ONLY the 3 decent GPU-offload MoE models. Small models were outclassed by these
|
|
# MoEs (35B-class knowledge at 3B active) and/or too slow to be useful on 4GB.
|
|
# ==============================================================================
|
|
|
|
healthCheckTimeout: 600 # 35B MoE mmap first-touch can take minutes
|
|
logLevel: info
|
|
startPort: 5800
|
|
|
|
macros:
|
|
# t=6 physical cores only — HT hurts (FINDINGS.md §5)
|
|
"server-base": >
|
|
/app/llama-server
|
|
--host 127.0.0.1 --port ${PORT}
|
|
--threads 6 --threads-batch 6
|
|
--flash-attn on
|
|
|
|
# upstream master build (newer MoE perf work); needs its own libs
|
|
"server-upstream": >
|
|
/app-upstream/llama-server
|
|
--host 127.0.0.1 --port ${PORT}
|
|
--threads 6 --threads-batch 6
|
|
--flash-attn on
|
|
|
|
# ik_llama.cpp: -ser/-fmoe/-rtr, fast IQ CPU kernels (flag names differ: -fa)
|
|
# needs env: LD_LIBRARY_PATH=/app-ik (own libllama/libggml, symbol-incompatible with turboquant's)
|
|
"server-ik": >
|
|
/app-ik/llama-server
|
|
--host 127.0.0.1 --port ${PORT}
|
|
--threads 6 --threads-batch 6 -fa on
|
|
|
|
"q8-kv": "--cache-type-k q8_0 --cache-type-v q8_0"
|
|
|
|
# MoE offload: dense backbone on GPU, routed experts in RAM/page cache.
|
|
# mmap (no mlock) mandatory for files > RAM.
|
|
"moe-offload": "--n-gpu-layers 99 --cpu-moe --jinja"
|
|
|
|
models:
|
|
|
|
# ── The 3 keepers: GPU-offload MoE (dense backbone on GPU, experts in RAM) ───
|
|
|
|
"ornith-35b":
|
|
name: "Ornith-1.0-35B Q2_K_L (GPU expert-offload, 128K)"
|
|
description: "MAIN coder/planner (pi solo). GPU expert-offload (ngl 99, n-cpu-moe 40), on-GPU KV, q8 K + turbo2 V. 28 t/s @ full 128K ctx. ~2.9GB VRAM. Requires nvidia_uvm un-wedged (see memory)."
|
|
cmd: |
|
|
${server-base}
|
|
--cache-type-k q8_0 --cache-type-v turbo2
|
|
--model /models/deepreinforce-ai_Ornith-1.0-35B-Q2_K_L.gguf
|
|
--n-gpu-layers 99 --n-cpu-moe 40 --jinja --reasoning-budget 512
|
|
--ctx-size 131072
|
|
--batch-size 1024 --ubatch-size 512
|
|
--cont-batching --parallel 1
|
|
--slot-save-path /cache/ornith --cache-reuse 256
|
|
ttl: 0
|
|
|
|
"qwen36-35b-q2":
|
|
name: "Qwen3.6-35B-A3B UD-Q2_K_XL"
|
|
description: "DAILY DRIVER 35B: 23.0 tg / 36 pp. ds4 asymmetric recipe (dense high-bit, experts 2-bit), 12.3GB page-cache resident, last 3 expert layers in VRAM. Gates pass"
|
|
cmd: |
|
|
${server-base} ${q8-kv}
|
|
--model /models/Qwen3.6-35B-A3B-UD-Q2_K_XL.gguf
|
|
--n-gpu-layers 99 --n-cpu-moe 37 --jinja
|
|
--ctx-size 32768
|
|
--batch-size 1024 --ubatch-size 512
|
|
--cont-batching --parallel 1
|
|
ttl: 300
|
|
|
|
"gpt-oss-20b":
|
|
name: "gpt-oss-20b MXFP4"
|
|
description: "20.9B/3.6B-active MoE, native MXFP4. 16.7 tg / 29.5 pp (n-cpu-moe 21: last 3 expert layers in VRAM, +16%)"
|
|
cmd: |
|
|
${server-base} ${q8-kv}
|
|
--model /models/gpt-oss-20b-mxfp4.gguf
|
|
--n-gpu-layers 99 --n-cpu-moe 21 --jinja
|
|
--ctx-size 32768
|
|
--batch-size 1024 --ubatch-size 512
|
|
--cont-batching --parallel 1
|
|
ttl: 300
|
|
|
|
# ── Bonsai-27B Q1_0 (1-bit ternary, PrismML fork) — fully on the 1650 Ti (4GB) ──
|
|
# 3.8GB weights ALL on GPU (ngl 99); KV in CPU RAM (--no-kv-offload) since VRAM
|
|
# can't hold both. q4_0 KV + --flash-attn = GPU attention (iq4_nl would force CPU).
|
|
# Tiny batch (-b 64) keeps the compute buffer under the ~48MB VRAM headroom.
|
|
# ctx is RAM-limited (KV offloaded), not VRAM. Trade: decode ~11 t/s (KV over
|
|
# PCIe), prefill ~420 pp. Binary: /app/llama-server-ternary (Q1_0_g128 kernels).
|
|
"bonsai-27b-q1":
|
|
name: "Bonsai-27B Q1_0 (1-bit ternary, all-GPU weights + KV in RAM)"
|
|
description: "Small fast local model. 27B ternary @ 1-bit, 3.8GB weights fully on the 1650 Ti; KV offloaded to RAM (q4_0). ~11 t/s decode, ~420 pp prefill. Hybrid-thinking -> reasoning_content; give >=800 max_tokens."
|
|
cmd: |
|
|
/app/llama-server-ternary
|
|
--host 127.0.0.1 --port ${PORT}
|
|
--threads 6 --threads-batch 6
|
|
--model /models/Bonsai-27B-Q1_0.gguf
|
|
--n-gpu-layers 99 --main-gpu 0
|
|
--flash-attn on
|
|
--cache-type-k q4_0 --cache-type-v q4_0
|
|
--no-kv-offload
|
|
--batch-size 64 --ubatch-size 64
|
|
--ctx-size 16384
|
|
--jinja
|
|
ttl: 300
|