Duo group (resident main+subagent for pi): flip VRAM to the small model. Ornith experts never touch VRAM; its dense-on-GPU split starved qwen to 176 MiB / 5.8 t/s concurrent. After flip: qwen 43 t/s, ornith 8 t/s. CUDA_VISIBLE_DEVICES= required for ornith — ngl 0 still allocates ~1GB pp compute buffer on CUDA builds (OOM+segfault). Duo section in MOE-FINDINGS.md; also snapshots prior swap-stack migration state. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
42 lines
1.9 KiB
Python
42 lines
1.9 KiB
Python
#!/usr/bin/env python3
|
|
"""Quick MoE bench via swap-stack. Usage: moe_bench.py <model> [n_runs]
|
|
Measures tg/pp from server .timings (never wall clock), plus a temp-0
|
|
structure gate (JSON validity + bracket balance). x570 method."""
|
|
import json, sys, time, urllib.request
|
|
|
|
BASE = "http://localhost:48080/v1/chat/completions"
|
|
MODEL = sys.argv[1] if len(sys.argv) > 1 else "gpt-oss-20b"
|
|
RUNS = int(sys.argv[2]) if len(sys.argv) > 2 else 2
|
|
|
|
def ask(prompt, mt=350):
|
|
req = urllib.request.Request(BASE,
|
|
data=json.dumps({"model": MODEL, "messages": [{"role": "user", "content": prompt}],
|
|
"max_tokens": mt, "temperature": 0}).encode(),
|
|
headers={"Content-Type": "application/json"})
|
|
r = json.load(urllib.request.urlopen(req, timeout=900))
|
|
t = r.get("timings", {})
|
|
return (r["choices"][0]["message"].get("content") or ""), t
|
|
|
|
# warmup / load
|
|
_, t0 = ask("Say OK.", mt=400)
|
|
print(f"warmup: tg={t0.get('predicted_per_second',0):.2f}")
|
|
|
|
# structure gate (temp 0). Generous budget: thinking models burn tokens on
|
|
# reasoning_content before emitting content (ornith needs ~250+).
|
|
c, _ = ask('Return JSON: {"name":"x","primes":[first 8 primes],"nested":{"a":true}}. JSON only.', mt=1500)
|
|
try:
|
|
json.loads(c); gate_json = "PASS"
|
|
except Exception:
|
|
gate_json = "FAIL: " + c[:120]
|
|
code, _ = ask("Write a Python function parsing nested brackets ()[]{} into a tree. Code only.", mt=2000)
|
|
gate_code = "PASS" if all(code.count(a) == code.count(b) for a, b in [("(",")"),("[","]"),("{","}")]) else "FAIL"
|
|
print(f"gate: json={gate_json} brackets={gate_code}")
|
|
|
|
# throughput runs
|
|
tgs, pps = [], []
|
|
for i in range(RUNS):
|
|
_, t = ask("Write a detailed essay about the history of computing.", mt=300)
|
|
tgs.append(t.get("predicted_per_second", 0)); pps.append(t.get("prompt_per_second", 0))
|
|
print(f"run{i+1}: tg={tgs[-1]:.2f} pp={pps[-1]:.2f}")
|
|
print(f"RESULT {MODEL}: tg_avg={sum(tgs)/len(tgs):.2f} pp_avg={sum(pps)/len(pps):.2f}")
|