diff --git a/compose.yaml b/compose.yaml index 8f1db21..b717a81 100644 --- a/compose.yaml +++ b/compose.yaml @@ -118,7 +118,7 @@ networks: # ── Volumes ─────────────────────────────────────────────────────────────────── volumes: - open-webui-data: + anythingllm-data: # ============================================================================== services: @@ -330,20 +330,27 @@ services: # docker compose --profile --profile webui up -d # Connects to whichever model is running via the llama-current DNS alias. # Open WebUI retries on startup so no depends_on needed. - openwebui: - image: ghcr.io/open-webui/open-webui:main - container_name: open_webui + # ── AnythingLLM (replaced open-webui 2026-07-10) ──────────────────────────── + # docker compose --profile webui up -d anythingllm + # Talks to the swap stack by service name; pick any model id from + # swap-stack/config.yaml (duo ids keep the resident pair loaded). + anythingllm: + image: mintplexlabs/anythingllm:latest + container_name: anythingllm profiles: [webui] environment: - - OPENAI_API_BASE_URL=http://llama-current:8080/v1 - - OPENAI_API_KEY=sk-no-key-needed - - WEBUI_AUTH=false + - LLM_PROVIDER=generic-openai + - GENERIC_OPEN_AI_BASE_PATH=http://llama-swap-stack:8080/v1 + - GENERIC_OPEN_AI_API_KEY=sk-no-key-needed + - GENERIC_OPEN_AI_MODEL_PREF=ornith-35b-duo + - GENERIC_OPEN_AI_MODEL_TOKEN_LIMIT=131072 + - STORAGE_DIR=/app/server/storage ports: - - "3000:8080" + - "3001:3001" networks: - llama-net volumes: - - open-webui-data:/app/backend/data + - anythingllm-data:/app/server/storage restart: unless-stopped # ── BENCHMARKS ───────────────────────────────────────────────────────────── diff --git a/swap-stack/config.yaml b/swap-stack/config.yaml index ed07dc3..4b4851c 100644 --- a/swap-stack/config.yaml +++ b/swap-stack/config.yaml @@ -102,17 +102,6 @@ models: --cont-batching --parallel 1 ttl: 300 - "qwen36-35b": - name: "Qwen3.6-35B-A3B UD-IQ4_XS" - description: "35B/3B-active MoE, 17.7GB mmap > RAM. Stop heavy containers first" - cmd: | - ${server-base} ${moe-offload} ${q8-kv} - --model /models/Qwen3.6-35B-A3B-UD-IQ4_XS.gguf - --ctx-size 32768 - --batch-size 1024 --ubatch-size 512 - --cont-batching --parallel 1 - ttl: 300 - "qwen36-35b-q2": name: "Qwen3.6-35B-A3B UD-Q2_K_XL" description: "DAILY DRIVER 35B: 23.0 tg / 36 pp. ds4 asymmetric recipe (dense high-bit, experts 2-bit), 12.3GB page-cache resident, last 3 expert layers in VRAM. Gates pass" @@ -137,16 +126,9 @@ models: --cont-batching --parallel 1 ttl: 300 - "ornith-35b-iq4": - name: "Ornith-1.0-35B IQ4_XS" - description: "Quality-first variant, 18.8GB mmap > RAM = ~2.8 t/s thrash. Batch jobs only (or post-RAM-upgrade)" - cmd: | - ${server-base} ${moe-offload} ${q8-kv} - --model /models/deepreinforce-ai_Ornith-1.0-35B-IQ4_XS.gguf - --ctx-size 32768 - --batch-size 1024 --ubatch-size 512 - --cont-batching --parallel 1 - ttl: 300 + # (IQ4 variants of ornith-35b/qwen36-35b deleted 2026-07-10 — 36.5GB disk, + # both were >RAM thrash-only until the 64GB upgrade. Re-download if needed: + # bartowski Ornith-1.0-35B IQ4_XS, unsloth Qwen3.6-35B-A3B UD-IQ4_XS.) # ── Dense 9B (RAM-bandwidth-bound, ~4.4 t/s) ─────────────────────────────── # (ik_llama / upstream-master A/B variants removed 2026-07-10 — questions