# Tri-binary swap-stack: llama-swap + three llama-server builds. # /app/llama-server TurboQuant fork (May 2026) — turbo2/3/4 KV, needed by 9B configs # /app-upstream/llama-server upstream master (Jul 2026) — newest MoE/arch work # /app/llama-server-ik ik_llama.cpp — -ser / -fmoe / -rtr, fast IQ-quant CPU kernels # /app/llama-server-ternary PrismML fork — Q1_0/Q2_0 g128 ternary kernels (Bonsai-27B) # config.yaml picks the binary per model (env: LD_LIBRARY_PATH for upstream). FROM local/llama-cpp-upstream:server-cuda-sm75-mmq AS upstream FROM local/ik-llama:server-cuda-sm75 AS ik # ── PrismML ternary build (Q1_0_g128 hybrid-attention kernels, sm75, static) ── # Bonsai-27B-Q1_0: 1-bit ternary, mainline/TheTom lack the g128 kernels. Built # STATIC so it can't collide with the other forks' libs; only needs the CUDA 12.8 # runtime libs present in the final turboquant image. USE q4_0 KV + --flash-attn # (NOT iq4_nl — CUDA FA doesn't support iq4_nl → CPU attention). Pinned 7529fdaaf. FROM nvidia/cuda:12.8.0-devel-ubuntu22.04 AS ternary RUN apt-get update && apt-get install -y --no-install-recommends \ git cmake ninja-build build-essential libcurl4-openssl-dev ca-certificates \ && rm -rf /var/lib/apt/lists/* RUN ln -sf /usr/local/cuda/lib64/stubs/libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1 \ && echo /usr/local/cuda/lib64/stubs > /etc/ld.so.conf.d/cuda-stubs.conf && ldconfig WORKDIR /src RUN git clone https://github.com/PrismML-Eng/llama.cpp.git . \ && git checkout 7529fdaaf && git log --oneline -1 RUN cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Release \ -DGGML_CUDA=ON -DGGML_NATIVE=OFF -DBUILD_SHARED_LIBS=OFF \ -DCMAKE_CUDA_ARCHITECTURES=75 -DGGML_CUDA_F16=ON -DLLAMA_CURL=ON \ -DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \ && cmake --build build --parallel "$(nproc)" --target llama-server FROM local/llama-cpp-turboquant:server-cuda-sm75-mmq COPY --from=upstream /app /app-upstream COPY --from=ik /llama-server /app-ik/llama-server COPY --from=ik /usr/local/lib/libllama.so /usr/local/lib/libggml.so /usr/local/lib/libmtmd.so /app-ik/ COPY --from=ik /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcudart.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublas.so.12 /usr/local/cuda-12.4/targets/x86_64-linux/lib/libcublasLt.so.12 /app-ik/ COPY --from=ternary /src/build/bin/llama-server /app/llama-server-ternary COPY llama-swap /app/llama-swap # config.yaml is bind-mounted at runtime (see compose.yaml) so edits # only need a container restart, not a rebuild. ENTRYPOINT ["/app/llama-swap", "-config", "/app/config.yaml", "-listen", ":8080"]