Files
llama-cpp/scripts/expert_heatmap.py
T
mozempkandClaude Fable 5 5e68d30d31 swap-stack: duo config — qwen3-4b full GPU, ornith-35b pure CPU
Duo group (resident main+subagent for pi): flip VRAM to the small model.
Ornith experts never touch VRAM; its dense-on-GPU split starved qwen to
176 MiB / 5.8 t/s concurrent. After flip: qwen 43 t/s, ornith 8 t/s.
CUDA_VISIBLE_DEVICES= required for ornith — ngl 0 still allocates ~1GB
pp compute buffer on CUDA builds (OOM+segfault). Duo section in
MOE-FINDINGS.md; also snapshots prior swap-stack migration state.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-10 08:28:32 +02:00

114 lines
5.1 KiB
Python

#!/usr/bin/env python3
"""Hot-expert mapper: page-cache residency per expert slice of a GGUF.
Expert tensors (blk.N.ffn_*_exps.weight) are fused 3D with the expert index
as the slowest dim -> each expert's weights are one contiguous byte range.
mincore() over each range after a real workload = which experts survived in
page cache (LRU proxy for routing heat). ds4/colibri hot-store idea, applied
externally to an unmodified llama.cpp.
Usage:
expert_heatmap.py <model.gguf> # snapshot + per-layer summary
expert_heatmap.py <model.gguf> --json out.json # full per-expert dump
"""
import ctypes, ctypes.util, json, mmap, os, struct, sys
libc = ctypes.CDLL(ctypes.util.find_library("c"), use_errno=True)
def read_gguf(path):
f = open(path, "rb")
assert f.read(4) == b"GGUF"
ver, = struct.unpack("<I", f.read(4))
n_tensors, = struct.unpack("<Q", f.read(8))
n_kv, = struct.unpack("<Q", f.read(8))
def rstr():
n, = struct.unpack("<Q", f.read(8)); return f.read(n).decode(errors="replace")
def rval(t):
fmt = {0:"<B",1:"<b",2:"<H",3:"<h",4:"<I",5:"<i",6:"<f",7:"<B",10:"<Q",11:"<q",12:"<d"}
if t == 8: return rstr()
if t == 9:
et, = struct.unpack("<I", f.read(4)); n, = struct.unpack("<Q", f.read(8))
return [rval(et) for _ in range(n)]
sz = struct.calcsize(fmt[t]); return struct.unpack(fmt[t], f.read(sz))[0]
kvs = {}
for _ in range(n_kv):
k = rstr(); t, = struct.unpack("<I", f.read(4)); v = rval(t)
kvs[k] = v if not isinstance(v, list) else None
tensors = []
for _ in range(n_tensors):
name = rstr(); nd, = struct.unpack("<I", f.read(4))
dims = struct.unpack(f"<{nd}Q", f.read(8 * nd))
ttype, = struct.unpack("<I", f.read(4)); off, = struct.unpack("<Q", f.read(8))
tensors.append([name, dims, ttype, off])
align = kvs.get("general.alignment") or 32
data_start = (f.tell() + align - 1) // align * align
f.close()
return kvs, tensors, data_start
def main():
path = sys.argv[1]
out_json = sys.argv[sys.argv.index("--json") + 1] if "--json" in sys.argv else None
kvs, tensors, data_start = read_gguf(path)
arch = kvs.get("general.architecture", "?")
n_expert = kvs.get(f"{arch}.expert_count") or 0
fsize = os.path.getsize(path)
# tensor byte sizes from offset deltas (robust across quant types)
tensors.sort(key=lambda t: t[3])
for i, t in enumerate(tensors):
nxt = tensors[i + 1][3] if i + 1 < len(tensors) else fsize - data_start
t.append(nxt - t[3]) # size
fd = os.open(path, os.O_RDONLY)
mm = mmap.mmap(fd, 0, prot=mmap.PROT_READ)
# read-only mmap: extract base address via the buffer protocol
class Py_buffer(ctypes.Structure):
_fields_ = [("buf", ctypes.c_void_p), ("obj", ctypes.py_object), ("len", ctypes.c_ssize_t),
("itemsize", ctypes.c_ssize_t), ("readonly", ctypes.c_int), ("ndim", ctypes.c_int),
("format", ctypes.c_char_p), ("shape", ctypes.c_void_p), ("strides", ctypes.c_void_p),
("suboffsets", ctypes.c_void_p), ("internal", ctypes.c_void_p)]
pybuf = Py_buffer()
ctypes.pythonapi.PyObject_GetBuffer(ctypes.py_object(mm), ctypes.byref(pybuf), ctypes.c_int(0))
addr = pybuf.buf
page = os.sysconf("SC_PAGESIZE")
def resident_fraction(off, size):
start = addr + off - (off % page)
length = size + (off % page)
npages = (length + page - 1) // page
vec = (ctypes.c_ubyte * npages)()
if libc.mincore(ctypes.c_void_p(start), ctypes.c_size_t(length), vec) != 0:
return -1.0
return sum(b & 1 for b in vec) / npages
layers = {} # layer -> {expert -> [frac,...] over up/gate/down}
for name, dims, ttype, off, size in tensors:
if "_exps.weight" not in name or n_expert == 0:
continue
layer = int(name.split(".")[1])
stride = size // n_expert
for e in range(n_expert):
frac = resident_fraction(data_start + off + e * stride, stride)
layers.setdefault(layer, {}).setdefault(e, []).append(frac)
print(f"# {os.path.basename(path)} arch={arch} experts/layer={n_expert} "
f"file={fsize/1e9:.1f}GB")
print(f"{'layer':>5} {'res%':>6} {'hot(>90%)':>9} {'cold(<10%)':>10} top5 experts")
summary = {}
for layer in sorted(layers):
em = {e: sum(v) / len(v) for e, v in layers[layer].items()}
avg = sum(em.values()) / len(em)
hot = sum(1 for v in em.values() if v > 0.9)
cold = sum(1 for v in em.values() if v < 0.1)
top = sorted(em, key=em.get, reverse=True)[:5]
summary[layer] = {"avg": avg, "hot": hot, "cold": cold, "experts": em}
print(f"{layer:>5} {avg*100:>5.1f}% {hot:>9} {cold:>10} {top}")
tot = [s["avg"] for s in summary.values()]
print(f"# overall expert residency: {sum(tot)/len(tot)*100:.1f}% "
f"(hottest layers: {sorted(summary, key=lambda l: summary[l]['avg'], reverse=True)[:6]})")
if out_json:
json.dump(summary, open(out_json, "w"))
print(f"# wrote {out_json}")
if __name__ == "__main__":
main()