swap-stack: drop concluded A/B build variants (ik, upstream-master)

FINDINGS answered their questions: ik_llama -23% decode / +13% prefill,
upstream master parity with turboquant on K-quant decode. Binaries stay
in the image, macros kept for future experiments.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-10 15:56:21 +02:00
co-authored by Claude Fable 5
parent 2a2305490a
commit aba5dd0a45
+3 -30
View File
@@ -148,37 +148,10 @@ models:
--cont-batching --parallel 1
ttl: 300
# ── ik_llama.cpp experimental variants ──────────────────────────────────────
"qwen36-35b-q2-ik":
name: "Qwen3.6-35B Q2_K_XL (ik_llama)"
description: "ik build: 18.0 tg / 42 pp — loses decode to main build (23.0), wins prefill. -ser 6,1 active. Experimental only"
env:
- "LD_LIBRARY_PATH=/app-ik"
cmd: |
${server-ik} ${q8-kv}
--model /models/Qwen3.6-35B-A3B-UD-Q2_K_XL.gguf
--n-gpu-layers 99 --n-cpu-moe 37 --jinja -ser 6,1
--ctx-size 16384
--batch-size 1024 --ubatch-size 512
--parallel 1
ttl: 300
"gpt-oss-20b-upstream":
name: "gpt-oss-20b (upstream master)"
description: "Upstream Jul-2026 build comparison"
env:
- "LD_LIBRARY_PATH=/app-upstream"
cmd: |
${server-upstream} ${q8-kv}
--model /models/gpt-oss-20b-mxfp4.gguf
--n-gpu-layers 99 --n-cpu-moe 21 --jinja
--ctx-size 32768
--batch-size 1024 --ubatch-size 512
--cont-batching --parallel 1
ttl: 300
# ── Dense 9B (RAM-bandwidth-bound, ~4.4 t/s) ───────────────────────────────
# (ik_llama / upstream-master A/B variants removed 2026-07-10 — questions
# answered in FINDINGS: ik −23% decode, upstream parity. Macros kept for
# future experiments; binaries still in the image.)
"ornith-9b":
name: "Ornith-1.0-9B Q8_0"