test(moe): byte-identity gates and synthetic tiny-moe generator

This commit is contained in:
Helldez 2026-07-10 18:18:19 +02:00
parent 5273af6487
commit 4b5a5b55d2
2 changed files with 214 additions and 0 deletions

127
scripts/make-tiny-moe.py Normal file
View file

@ -0,0 +1,127 @@
#!/usr/bin/env python3
"""Generate a tiny random-weight qwen3moe gguf for the byte-identity gates.
The gates compare STREAMED-experts output against FULL-RESIDENT output of the SAME
file, so random weights are fine — quality is irrelevant, only that routing is a valid
top-k distribution (argsort of random logits) and that llama.cpp loads the model as a
MoE. The model is deliberately multi-layer with a few experts so the LRU cache path
sees real evictions on a small budget.
Requires: pip install gguf numpy
python scripts/make-tiny-moe.py --out tiny-moe.gguf
"""
import argparse
import numpy as np
try:
import gguf
except ImportError:
raise SystemExit("missing dependency: pip install gguf numpy")
# --- tiny architecture -------------------------------------------------------------
# Sized so the experts total a few MiB across layers: a small LRU budget (a couple MiB)
# then forces real evictions, exercising that path in the gates.
N_LAYER = 4
N_EMBD = 128
N_HEAD = 4
N_HEAD_KV = 2
N_EMBD_HEAD = N_EMBD // N_HEAD # 32
N_EMBD_GQA = N_HEAD_KV * N_EMBD_HEAD # 64
N_FF = 256
N_EXPERT = 8
N_EXPERT_USED = 2
N_FF_EXP = 128
N_CTX = 256
RMS_EPS = 1e-6
ROPE_BASE = 1000000.0
def build_vocab():
"""Minimal SPM byte-fallback vocab: 3 specials + 256 byte tokens."""
tokens, scores, toktypes = [], [], []
for t, ty in (("<unk>", gguf.TokenType.UNKNOWN),
("<s>", gguf.TokenType.CONTROL),
("</s>", gguf.TokenType.CONTROL)):
tokens.append(t); scores.append(0.0); toktypes.append(ty)
for b in range(256):
tokens.append(f"<0x{b:02X}>"); scores.append(0.0); toktypes.append(gguf.TokenType.BYTE)
return tokens, scores, toktypes
def rnd(*shape, seed):
g = np.random.default_rng(seed)
return g.standard_normal(shape).astype(np.float32) * 0.02
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--out", default="tiny-moe.gguf")
args = ap.parse_args()
tokens, scores, toktypes = build_vocab()
n_vocab = len(tokens)
w = gguf.GGUFWriter(args.out, "qwen3moe")
w.add_name("tiny-moe")
w.add_context_length(N_CTX)
w.add_embedding_length(N_EMBD)
w.add_block_count(N_LAYER)
w.add_feed_forward_length(N_FF)
w.add_head_count(N_HEAD)
w.add_head_count_kv(N_HEAD_KV)
w.add_key_length(N_EMBD_HEAD)
w.add_value_length(N_EMBD_HEAD)
w.add_rope_freq_base(ROPE_BASE)
w.add_layer_norm_rms_eps(RMS_EPS)
w.add_expert_count(N_EXPERT)
w.add_expert_used_count(N_EXPERT_USED)
w.add_expert_feed_forward_length(N_FF_EXP)
w.add_file_type(gguf.LlamaFileType.ALL_F32)
# tokenizer
w.add_tokenizer_model("llama")
w.add_tokenizer_pre("default")
w.add_token_list(tokens)
w.add_token_scores(scores)
w.add_token_types(toktypes)
w.add_unk_token_id(0)
w.add_bos_token_id(1)
w.add_eos_token_id(2)
w.add_add_bos_token(True)
w.add_add_eos_token(False)
# global tensors (numpy shapes are ggml dims reversed)
w.add_tensor("token_embd.weight", rnd(n_vocab, N_EMBD, seed=1))
w.add_tensor("output_norm.weight", rnd(N_EMBD, seed=2))
w.add_tensor("output.weight", rnd(n_vocab, N_EMBD, seed=3))
s = 100
for i in range(N_LAYER):
p = f"blk.{i}."
w.add_tensor(p + "attn_norm.weight", rnd(N_EMBD, seed=s + 0))
w.add_tensor(p + "attn_q.weight", rnd(N_EMBD_HEAD * N_HEAD, N_EMBD, seed=s + 1))
w.add_tensor(p + "attn_k.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 2))
w.add_tensor(p + "attn_v.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 3))
w.add_tensor(p + "attn_output.weight", rnd(N_EMBD, N_EMBD_HEAD * N_HEAD, seed=s + 4))
w.add_tensor(p + "attn_q_norm.weight", rnd(N_EMBD_HEAD, seed=s + 5))
w.add_tensor(p + "attn_k_norm.weight", rnd(N_EMBD_HEAD, seed=s + 6))
w.add_tensor(p + "ffn_norm.weight", rnd(N_EMBD, seed=s + 7))
w.add_tensor(p + "ffn_gate_inp.weight", rnd(N_EXPERT, N_EMBD, seed=s + 8))
# experts: dim-2 (numpy axis 0) indexes the expert
w.add_tensor(p + "ffn_gate_exps.weight", rnd(N_EXPERT, N_FF_EXP, N_EMBD, seed=s + 9))
w.add_tensor(p + "ffn_down_exps.weight", rnd(N_EXPERT, N_EMBD, N_FF_EXP, seed=s + 10))
w.add_tensor(p + "ffn_up_exps.weight", rnd(N_EXPERT, N_FF_EXP, N_EMBD, seed=s + 11))
s += 100
w.write_header_to_file()
w.write_kv_data_to_file()
w.write_tensors_to_file()
w.close()
print(f"wrote {args.out}: qwen3moe, {N_LAYER} layers, {N_EXPERT} experts "
f"(top-{N_EXPERT_USED}), vocab {n_vocab}")
if __name__ == "__main__":
main()

87
tests/moe_gates.cpp Normal file
View file

@ -0,0 +1,87 @@
// Byte-identity gates for MoE expert streaming.
//
// Greedy generation is a deterministic function of the graph, so streaming only the
// routed experts must produce output identical to running with every expert resident.
// These gates assert exactly that on the tiny synthetic model (scripts/make-tiny-moe.py),
// which the test harness generates first. Pass the model path as argv[1].
//
// G1 resident (no streaming) == streaming, cache off
// G2 streaming, cache off == streaming, small LRU cache (forces evictions)
// G3 streaming, selective == streaming, --load-all (every expert each token)
//
// If G3 passes, the streamer provably never gathers an unrouted (garbage) slice.
#include "bmoe/config.h"
#include "bmoe/runtime.h"
#include <cstdio>
#include <string>
using namespace bmoe;
static RunConfig base(const std::string & model) {
RunConfig c;
c.model_path = model;
c.prompt = "Hello world, this is a streaming test.";
c.n_predict = 24;
c.n_threads = 2;
c.n_ctx = 256;
return c;
}
static bool gen(const RunConfig & c, std::string & out, std::string & err) {
RunResult r = run(c);
if (!r) { err = r.error; return false; }
out = r.generated_text;
return true;
}
static int check(const char * name, const std::string & a, const std::string & b) {
if (a == b) {
std::printf("[PASS] %s\n", name);
return 0;
}
std::printf("[FAIL] %s\n A: %s\n B: %s\n", name, a.c_str(), b.c_str());
return 1;
}
int main(int argc, char ** argv) {
if (argc < 2) { std::fprintf(stderr, "usage: %s <tiny-moe.gguf>\n", argv[0]); return 2; }
const std::string model = argv[1];
// resident reference
RunConfig resident = base(model);
resident.moe.enabled = false;
// streaming, cache off
RunConfig stream0 = base(model);
stream0.moe.enabled = true;
stream0.moe.cache_mb = 0;
stream0.moe.io_threads = 4;
// streaming, small LRU cache (pathological band → force it on for the test)
RunConfig streamc = base(model);
streamc.moe.enabled = true;
streamc.moe.cache_mb = 2;
streamc.moe.force_cache = true;
streamc.moe.io_threads = 4;
// streaming, load-all baseline
RunConfig streamall = base(model);
streamall.moe.enabled = true;
streamall.moe.cache_mb = 0;
streamall.moe.load_all = true;
std::string s_res, s_s0, s_sc, s_all, err;
if (!gen(resident, s_res, err)) { std::fprintf(stderr, "resident run failed: %s\n", err.c_str()); return 2; }
if (!gen(stream0, s_s0, err)) { std::fprintf(stderr, "stream0 run failed: %s\n", err.c_str()); return 2; }
if (!gen(streamc, s_sc, err)) { std::fprintf(stderr, "streamc run failed: %s\n", err.c_str()); return 2; }
if (!gen(streamall, s_all, err)) { std::fprintf(stderr, "load-all run failed: %s\n", err.c_str()); return 2; }
int fails = 0;
fails += check("G1 resident == streaming(cache off)", s_res, s_s0);
fails += check("G2 streaming(cache off) == streaming(LRU cache)", s_s0, s_sc);
fails += check("G3 streaming(selective) == streaming(load-all)", s_s0, s_all);
if (fails == 0) std::printf("\nall MoE byte-identity gates passed\n");
return fails == 0 ? 0 : 1;
}