mirror of
https://github.com/Helldez/BigMoeOnEdge.git
synced 2026-10-03 03:25:42 +00:00
test(moe): byte-identity gates and synthetic tiny-moe generator
This commit is contained in:
parent
5273af6487
commit
4b5a5b55d2
2 changed files with 214 additions and 0 deletions
127
scripts/make-tiny-moe.py
Normal file
127
scripts/make-tiny-moe.py
Normal file
|
|
@ -0,0 +1,127 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Generate a tiny random-weight qwen3moe gguf for the byte-identity gates.
|
||||
|
||||
The gates compare STREAMED-experts output against FULL-RESIDENT output of the SAME
|
||||
file, so random weights are fine — quality is irrelevant, only that routing is a valid
|
||||
top-k distribution (argsort of random logits) and that llama.cpp loads the model as a
|
||||
MoE. The model is deliberately multi-layer with a few experts so the LRU cache path
|
||||
sees real evictions on a small budget.
|
||||
|
||||
Requires: pip install gguf numpy
|
||||
|
||||
python scripts/make-tiny-moe.py --out tiny-moe.gguf
|
||||
"""
|
||||
import argparse
|
||||
import numpy as np
|
||||
|
||||
try:
|
||||
import gguf
|
||||
except ImportError:
|
||||
raise SystemExit("missing dependency: pip install gguf numpy")
|
||||
|
||||
# --- tiny architecture -------------------------------------------------------------
|
||||
# Sized so the experts total a few MiB across layers: a small LRU budget (a couple MiB)
|
||||
# then forces real evictions, exercising that path in the gates.
|
||||
N_LAYER = 4
|
||||
N_EMBD = 128
|
||||
N_HEAD = 4
|
||||
N_HEAD_KV = 2
|
||||
N_EMBD_HEAD = N_EMBD // N_HEAD # 32
|
||||
N_EMBD_GQA = N_HEAD_KV * N_EMBD_HEAD # 64
|
||||
N_FF = 256
|
||||
N_EXPERT = 8
|
||||
N_EXPERT_USED = 2
|
||||
N_FF_EXP = 128
|
||||
N_CTX = 256
|
||||
RMS_EPS = 1e-6
|
||||
ROPE_BASE = 1000000.0
|
||||
|
||||
|
||||
def build_vocab():
|
||||
"""Minimal SPM byte-fallback vocab: 3 specials + 256 byte tokens."""
|
||||
tokens, scores, toktypes = [], [], []
|
||||
for t, ty in (("<unk>", gguf.TokenType.UNKNOWN),
|
||||
("<s>", gguf.TokenType.CONTROL),
|
||||
("</s>", gguf.TokenType.CONTROL)):
|
||||
tokens.append(t); scores.append(0.0); toktypes.append(ty)
|
||||
for b in range(256):
|
||||
tokens.append(f"<0x{b:02X}>"); scores.append(0.0); toktypes.append(gguf.TokenType.BYTE)
|
||||
return tokens, scores, toktypes
|
||||
|
||||
|
||||
def rnd(*shape, seed):
|
||||
g = np.random.default_rng(seed)
|
||||
return g.standard_normal(shape).astype(np.float32) * 0.02
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--out", default="tiny-moe.gguf")
|
||||
args = ap.parse_args()
|
||||
|
||||
tokens, scores, toktypes = build_vocab()
|
||||
n_vocab = len(tokens)
|
||||
|
||||
w = gguf.GGUFWriter(args.out, "qwen3moe")
|
||||
|
||||
w.add_name("tiny-moe")
|
||||
w.add_context_length(N_CTX)
|
||||
w.add_embedding_length(N_EMBD)
|
||||
w.add_block_count(N_LAYER)
|
||||
w.add_feed_forward_length(N_FF)
|
||||
w.add_head_count(N_HEAD)
|
||||
w.add_head_count_kv(N_HEAD_KV)
|
||||
w.add_key_length(N_EMBD_HEAD)
|
||||
w.add_value_length(N_EMBD_HEAD)
|
||||
w.add_rope_freq_base(ROPE_BASE)
|
||||
w.add_layer_norm_rms_eps(RMS_EPS)
|
||||
w.add_expert_count(N_EXPERT)
|
||||
w.add_expert_used_count(N_EXPERT_USED)
|
||||
w.add_expert_feed_forward_length(N_FF_EXP)
|
||||
w.add_file_type(gguf.LlamaFileType.ALL_F32)
|
||||
|
||||
# tokenizer
|
||||
w.add_tokenizer_model("llama")
|
||||
w.add_tokenizer_pre("default")
|
||||
w.add_token_list(tokens)
|
||||
w.add_token_scores(scores)
|
||||
w.add_token_types(toktypes)
|
||||
w.add_unk_token_id(0)
|
||||
w.add_bos_token_id(1)
|
||||
w.add_eos_token_id(2)
|
||||
w.add_add_bos_token(True)
|
||||
w.add_add_eos_token(False)
|
||||
|
||||
# global tensors (numpy shapes are ggml dims reversed)
|
||||
w.add_tensor("token_embd.weight", rnd(n_vocab, N_EMBD, seed=1))
|
||||
w.add_tensor("output_norm.weight", rnd(N_EMBD, seed=2))
|
||||
w.add_tensor("output.weight", rnd(n_vocab, N_EMBD, seed=3))
|
||||
|
||||
s = 100
|
||||
for i in range(N_LAYER):
|
||||
p = f"blk.{i}."
|
||||
w.add_tensor(p + "attn_norm.weight", rnd(N_EMBD, seed=s + 0))
|
||||
w.add_tensor(p + "attn_q.weight", rnd(N_EMBD_HEAD * N_HEAD, N_EMBD, seed=s + 1))
|
||||
w.add_tensor(p + "attn_k.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 2))
|
||||
w.add_tensor(p + "attn_v.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 3))
|
||||
w.add_tensor(p + "attn_output.weight", rnd(N_EMBD, N_EMBD_HEAD * N_HEAD, seed=s + 4))
|
||||
w.add_tensor(p + "attn_q_norm.weight", rnd(N_EMBD_HEAD, seed=s + 5))
|
||||
w.add_tensor(p + "attn_k_norm.weight", rnd(N_EMBD_HEAD, seed=s + 6))
|
||||
w.add_tensor(p + "ffn_norm.weight", rnd(N_EMBD, seed=s + 7))
|
||||
w.add_tensor(p + "ffn_gate_inp.weight", rnd(N_EXPERT, N_EMBD, seed=s + 8))
|
||||
# experts: dim-2 (numpy axis 0) indexes the expert
|
||||
w.add_tensor(p + "ffn_gate_exps.weight", rnd(N_EXPERT, N_FF_EXP, N_EMBD, seed=s + 9))
|
||||
w.add_tensor(p + "ffn_down_exps.weight", rnd(N_EXPERT, N_EMBD, N_FF_EXP, seed=s + 10))
|
||||
w.add_tensor(p + "ffn_up_exps.weight", rnd(N_EXPERT, N_FF_EXP, N_EMBD, seed=s + 11))
|
||||
s += 100
|
||||
|
||||
w.write_header_to_file()
|
||||
w.write_kv_data_to_file()
|
||||
w.write_tensors_to_file()
|
||||
w.close()
|
||||
print(f"wrote {args.out}: qwen3moe, {N_LAYER} layers, {N_EXPERT} experts "
|
||||
f"(top-{N_EXPERT_USED}), vocab {n_vocab}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
87
tests/moe_gates.cpp
Normal file
87
tests/moe_gates.cpp
Normal file
|
|
@ -0,0 +1,87 @@
|
|||
// Byte-identity gates for MoE expert streaming.
|
||||
//
|
||||
// Greedy generation is a deterministic function of the graph, so streaming only the
|
||||
// routed experts must produce output identical to running with every expert resident.
|
||||
// These gates assert exactly that on the tiny synthetic model (scripts/make-tiny-moe.py),
|
||||
// which the test harness generates first. Pass the model path as argv[1].
|
||||
//
|
||||
// G1 resident (no streaming) == streaming, cache off
|
||||
// G2 streaming, cache off == streaming, small LRU cache (forces evictions)
|
||||
// G3 streaming, selective == streaming, --load-all (every expert each token)
|
||||
//
|
||||
// If G3 passes, the streamer provably never gathers an unrouted (garbage) slice.
|
||||
#include "bmoe/config.h"
|
||||
#include "bmoe/runtime.h"
|
||||
|
||||
#include <cstdio>
|
||||
#include <string>
|
||||
|
||||
using namespace bmoe;
|
||||
|
||||
static RunConfig base(const std::string & model) {
|
||||
RunConfig c;
|
||||
c.model_path = model;
|
||||
c.prompt = "Hello world, this is a streaming test.";
|
||||
c.n_predict = 24;
|
||||
c.n_threads = 2;
|
||||
c.n_ctx = 256;
|
||||
return c;
|
||||
}
|
||||
|
||||
static bool gen(const RunConfig & c, std::string & out, std::string & err) {
|
||||
RunResult r = run(c);
|
||||
if (!r) { err = r.error; return false; }
|
||||
out = r.generated_text;
|
||||
return true;
|
||||
}
|
||||
|
||||
static int check(const char * name, const std::string & a, const std::string & b) {
|
||||
if (a == b) {
|
||||
std::printf("[PASS] %s\n", name);
|
||||
return 0;
|
||||
}
|
||||
std::printf("[FAIL] %s\n A: %s\n B: %s\n", name, a.c_str(), b.c_str());
|
||||
return 1;
|
||||
}
|
||||
|
||||
int main(int argc, char ** argv) {
|
||||
if (argc < 2) { std::fprintf(stderr, "usage: %s <tiny-moe.gguf>\n", argv[0]); return 2; }
|
||||
const std::string model = argv[1];
|
||||
|
||||
// resident reference
|
||||
RunConfig resident = base(model);
|
||||
resident.moe.enabled = false;
|
||||
|
||||
// streaming, cache off
|
||||
RunConfig stream0 = base(model);
|
||||
stream0.moe.enabled = true;
|
||||
stream0.moe.cache_mb = 0;
|
||||
stream0.moe.io_threads = 4;
|
||||
|
||||
// streaming, small LRU cache (pathological band → force it on for the test)
|
||||
RunConfig streamc = base(model);
|
||||
streamc.moe.enabled = true;
|
||||
streamc.moe.cache_mb = 2;
|
||||
streamc.moe.force_cache = true;
|
||||
streamc.moe.io_threads = 4;
|
||||
|
||||
// streaming, load-all baseline
|
||||
RunConfig streamall = base(model);
|
||||
streamall.moe.enabled = true;
|
||||
streamall.moe.cache_mb = 0;
|
||||
streamall.moe.load_all = true;
|
||||
|
||||
std::string s_res, s_s0, s_sc, s_all, err;
|
||||
if (!gen(resident, s_res, err)) { std::fprintf(stderr, "resident run failed: %s\n", err.c_str()); return 2; }
|
||||
if (!gen(stream0, s_s0, err)) { std::fprintf(stderr, "stream0 run failed: %s\n", err.c_str()); return 2; }
|
||||
if (!gen(streamc, s_sc, err)) { std::fprintf(stderr, "streamc run failed: %s\n", err.c_str()); return 2; }
|
||||
if (!gen(streamall, s_all, err)) { std::fprintf(stderr, "load-all run failed: %s\n", err.c_str()); return 2; }
|
||||
|
||||
int fails = 0;
|
||||
fails += check("G1 resident == streaming(cache off)", s_res, s_s0);
|
||||
fails += check("G2 streaming(cache off) == streaming(LRU cache)", s_s0, s_sc);
|
||||
fails += check("G3 streaming(selective) == streaming(load-all)", s_s0, s_all);
|
||||
|
||||
if (fails == 0) std::printf("\nall MoE byte-identity gates passed\n");
|
||||
return fails == 0 ? 0 : 1;
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue