Add a 4th binary to the swap-stack: PrismML llama-server-ternary (Q1_0_g128 ternary kernels, built sm75/CUDA-12.8 static) for Bonsai-27B-Q1_0. Config: 3.8GB weights ALL on the 4GB GPU (ngl 99) + KV in CPU RAM (--no-kv-offload, since VRAM cannot hold both) + q4_0 KV + --flash-attn (GPU attention; iq4_nl forces CPU attention) + tiny batch (-b 64) to fit the ~100MB VRAM headroom. Fits at 3611/3718 MiB. ~7 t/s decode, ~27 pp prefill (1650 Ti has no tensor cores + KV over PCIe = the ceiling). Measured alt configs (all worse): ngl48+KV-VRAM = 8.5 t/s decode but only 10 pp prefill (CPU layers tank prefill); ngl99+KV-VRAM OOMs even at ctx 2048. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
42 lines
2.6 KiB
Docker
42 lines
2.6 KiB
Docker
# Tri-binary swap-stack: llama-swap + three llama-server builds.
|
|
# /app/llama-server TurboQuant fork (May 2026) — turbo2/3/4 KV, needed by 9B configs
|
|
# /app-upstream/llama-server upstream master (Jul 2026) — newest MoE/arch work
|
|
# /app/llama-server-ik ik_llama.cpp — -ser / -fmoe / -rtr, fast IQ-quant CPU kernels
|
|
# /app/llama-server-ternary PrismML fork — Q1_0/Q2_0 g128 ternary kernels (Bonsai-27B)
|
|
# config.yaml picks the binary per model (env: LD_LIBRARY_PATH for upstream).
|
|
FROM local/llama-cpp-upstream:server-cuda-sm75-mmq AS upstream
|
|
FROM local/ik-llama:server-cuda-sm75 AS ik
|
|
|
|
# ── PrismML ternary build (Q1_0_g128 hybrid-attention kernels, sm75, static) ──
|
|
# Bonsai-27B-Q1_0: 1-bit ternary, mainline/TheTom lack the g128 kernels. Built
|
|
# STATIC so it can't collide with the other forks' libs; only needs the CUDA 12.8
|
|
# runtime libs present in the final turboquant image. USE q4_0 KV + --flash-attn
|
|
# (NOT iq4_nl — CUDA FA doesn't support iq4_nl → CPU attention). Pinned 7529fdaaf.
|
|
FROM nvidia/cuda:12.8.0-devel-ubuntu22.04 AS ternary
|
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
git cmake ninja-build build-essential libcurl4-openssl-dev ca-certificates \
|
|
&& rm -rf /var/lib/apt/lists/*
|
|
RUN ln -sf /usr/local/cuda/lib64/stubs/libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1 \
|
|
&& echo /usr/local/cuda/lib64/stubs > /etc/ld.so.conf.d/cuda-stubs.conf && ldconfig
|
|
WORKDIR /src
|
|
RUN git clone https://github.com/PrismML-Eng/llama.cpp.git . \
|
|
&& git checkout 7529fdaaf && git log --oneline -1
|
|
RUN cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release \
|
|
-DGGML_CUDA=ON -DGGML_NATIVE=OFF -DBUILD_SHARED_LIBS=OFF \
|
|
-DCMAKE_CUDA_ARCHITECTURES=75 -DGGML_CUDA_F16=ON -DLLAMA_CURL=ON \
|
|
-DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \
|
|
&& cmake --build build --parallel "$(nproc)" --target llama-server
|
|
|
|
FROM local/llama-cpp-turboquant:server-cuda-sm75-mmq
|
|
|
|
COPY --from=upstream /app /app-upstream
|
|
COPY --from=ik /llama-server /app-ik/llama-server
|
|
COPY --from=ik /usr/local/lib/libllama.so /usr/local/lib/libggml.so /usr/local/lib/libmtmd.so /app-ik/
|
|
COPY --from=ik /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcudart.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublas.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublasLt.so.12 /app-ik/
|
|
COPY --from=ternary /src/build/bin/llama-server /app/llama-server-ternary
|
|
COPY llama-swap /app/llama-swap
|
|
|
|
# config.yaml is bind-mounted at runtime (see compose.yaml) so edits
|
|
# only need a container restart, not a rebuild.
|
|
ENTRYPOINT ["/app/llama-swap", "-config", "/app/config.yaml", "-listen", ":8080"]
|