BigMoeOnEdge/scripts/make-tiny-moe.py
Raffaele 1d37057fac
Some checks failed
ci / changes (push) Has been cancelled
ci / host-windows (push) Has been cancelled
ci / host-macos (push) Has been cancelled
ci / android-apk (push) Has been cancelled
ci / format (push) Has been cancelled
ci / host-linux (push) Has been cancelled
feat: --decide, choose from a list from one prefill, with no decode (#201)
A session opened with --decide answers which of a list of choices the model would pick, read from
the next-token distribution after one prefill, with no decode. The state after a shared prefix is
kept and restored when the next prefix extends it. Android app: a Choose from options switch.

With --prefill-device, a decision is prefilled by the chat turn's placement rule and keeps no
prefix state: llama.cpp saves a sequence through KV views that do not follow the moved model state
(gate G18g). Also fixes the engine version, stuck at 0.23.0 since 0.24.0. App 0.27.0 (42).
2026-09-28 12:01:25 +02:00

401 lines
19 KiB
Python

#!/usr/bin/env python3
"""Generate a tiny random-weight MoE gguf for the byte-identity gates.
The gates compare STREAMED-experts output against FULL-RESIDENT output of the SAME
file, so random weights are fine — quality is irrelevant, only that routing is a valid
top-k distribution (argsort of random logits) and that llama.cpp loads the model as a
MoE. The model is deliberately multi-layer with a few experts so the LRU cache path
sees real evictions on a small budget.
Three architectures are emitted, selected with --arch:
qwen3moe split expert layout — three tensors per layer
(ffn_gate_exps / ffn_up_exps / ffn_down_exps).
gemma4 fused gate+up layout — two expert tensors per layer
(ffn_gate_up_exps / ffn_down_exps), plus a resident shared expert and an
interleaved dense layer, so the gates cover the fused streaming path.
nemotron_h_moe gate-less layout: two expert tensors per layer
(ffn_up_exps / ffn_down_exps, ReLU^2) in a hybrid Mamba2 / attention / MoE
stack with latent projections, a resident shared expert and a trailing MTP
block that is never loaded, so the gates cover the up+down streaming path.
Requires: pip install gguf numpy
python scripts/make-tiny-moe.py --arch qwen3moe --out tiny-moe.gguf
python scripts/make-tiny-moe.py --arch gemma4 --out tiny-moe-gemma4.gguf
python scripts/make-tiny-moe.py --arch nemotron_h_moe --out tiny-moe-nemotron_h_moe.gguf
"""
import argparse
import numpy as np
try:
import gguf
except ImportError:
raise SystemExit("missing dependency: pip install gguf numpy")
# --- tiny architecture -------------------------------------------------------------
# Sized so the experts total a few MiB across layers: a small LRU budget (a couple MiB)
# then forces real evictions, exercising that path in the gates.
N_LAYER = 4
N_EMBD = 128
N_HEAD = 4
N_HEAD_KV = 2
N_EMBD_HEAD = N_EMBD // N_HEAD # 32
N_EMBD_GQA = N_HEAD_KV * N_EMBD_HEAD # 64
N_FF = 256
N_EXPERT = 8
N_EXPERT_USED = 2
N_FF_EXP = 128
N_CTX = 256
RMS_EPS = 1e-6
ROPE_BASE = 1000000.0
def build_vocab():
"""Minimal SPM byte-fallback vocab: 3 specials + 256 byte tokens."""
tokens, scores, toktypes = [], [], []
for t, ty in (("<unk>", gguf.TokenType.UNKNOWN),
("<s>", gguf.TokenType.CONTROL),
("</s>", gguf.TokenType.CONTROL)):
tokens.append(t); scores.append(0.0); toktypes.append(ty)
for b in range(256):
tokens.append(f"<0x{b:02X}>"); scores.append(0.0); toktypes.append(gguf.TokenType.BYTE)
return tokens, scores, toktypes
def rnd(*shape, seed):
g = np.random.default_rng(seed)
return g.standard_normal(shape).astype(np.float32) * 0.02
def add_tokenizer(w, tokens, scores, toktypes):
w.add_tokenizer_model("llama")
w.add_tokenizer_pre("default")
w.add_token_list(tokens)
w.add_token_scores(scores)
w.add_token_types(toktypes)
w.add_unk_token_id(0)
w.add_bos_token_id(1)
w.add_eos_token_id(2)
w.add_add_bos_token(True)
w.add_add_eos_token(False)
# SPM prepends a space by default, and a byte-only vocab spells it as the three bytes of U+2581,
# so every text would start with the same token. The decide gates score choices by their first
# token, and "A", "B", "C" must stay distinguishable. Every other gate compares runs on the same
# model, so the setting is arbitrary for them.
w.add_add_space_prefix(False)
def add_attn_tensors(w, p, s):
"""Attention block shared by both architectures (numpy shapes are ggml dims reversed)."""
w.add_tensor(p + "attn_q.weight", rnd(N_EMBD_HEAD * N_HEAD, N_EMBD, seed=s + 1))
w.add_tensor(p + "attn_k.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 2))
w.add_tensor(p + "attn_v.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 3))
w.add_tensor(p + "attn_output.weight", rnd(N_EMBD, N_EMBD_HEAD * N_HEAD, seed=s + 4))
w.add_tensor(p + "attn_q_norm.weight", rnd(N_EMBD_HEAD, seed=s + 5))
w.add_tensor(p + "attn_k_norm.weight", rnd(N_EMBD_HEAD, seed=s + 6))
# --- qwen3moe: split expert layout -------------------------------------------------
def make_writer(out, arch, split_max_tensors):
"""A plain writer, or a sharding one when --split-max-tensors is set.
Sharded output mirrors how real >50 GB models arrive from Hugging Face: the writer
emits <out>-%05d-of-%05d.gguf siblings, with a metadata-only first shard
(small_first_shard, the layout unsloth ships). The byte-identity gates then prove the
multi-shard streaming path against the same tensors the single-file fixture uses.
"""
if not split_max_tensors:
return gguf.GGUFWriter(out, arch)
try:
return gguf.GGUFWriter(out, arch,
split_max_tensors=split_max_tensors,
small_first_shard=True)
except TypeError:
raise SystemExit("this gguf package cannot write split files: pip install -U gguf")
def build_qwen3moe(out, split_max_tensors=0):
tokens, scores, toktypes = build_vocab()
n_vocab = len(tokens)
w = make_writer(out, "qwen3moe", split_max_tensors)
w.add_name("tiny-moe")
w.add_context_length(N_CTX)
w.add_embedding_length(N_EMBD)
w.add_block_count(N_LAYER)
w.add_feed_forward_length(N_FF)
w.add_head_count(N_HEAD)
w.add_head_count_kv(N_HEAD_KV)
w.add_key_length(N_EMBD_HEAD)
w.add_value_length(N_EMBD_HEAD)
w.add_rope_freq_base(ROPE_BASE)
w.add_layer_norm_rms_eps(RMS_EPS)
w.add_expert_count(N_EXPERT)
w.add_expert_used_count(N_EXPERT_USED)
w.add_expert_feed_forward_length(N_FF_EXP)
w.add_file_type(gguf.LlamaFileType.ALL_F32)
add_tokenizer(w, tokens, scores, toktypes)
w.add_tensor("token_embd.weight", rnd(n_vocab, N_EMBD, seed=1))
w.add_tensor("output_norm.weight", rnd(N_EMBD, seed=2))
w.add_tensor("output.weight", rnd(n_vocab, N_EMBD, seed=3))
s = 100
for i in range(N_LAYER):
p = f"blk.{i}."
w.add_tensor(p + "attn_norm.weight", rnd(N_EMBD, seed=s + 0))
add_attn_tensors(w, p, s)
w.add_tensor(p + "ffn_norm.weight", rnd(N_EMBD, seed=s + 7))
w.add_tensor(p + "ffn_gate_inp.weight", rnd(N_EXPERT, N_EMBD, seed=s + 8))
# experts: dim-2 (numpy axis 0) indexes the expert
w.add_tensor(p + "ffn_gate_exps.weight", rnd(N_EXPERT, N_FF_EXP, N_EMBD, seed=s + 9))
w.add_tensor(p + "ffn_down_exps.weight", rnd(N_EXPERT, N_EMBD, N_FF_EXP, seed=s + 10))
w.add_tensor(p + "ffn_up_exps.weight", rnd(N_EXPERT, N_FF_EXP, N_EMBD, seed=s + 11))
s += 100
w.write_header_to_file()
w.write_kv_data_to_file()
w.write_tensors_to_file()
w.close()
print(f"wrote {out}: qwen3moe, {N_LAYER} layers, {N_EXPERT} experts "
f"(top-{N_EXPERT_USED}), vocab {n_vocab}")
# --- gemma4: fused gate+up layout --------------------------------------------------
# Gemma 4 MoE packs gate+up into one expert tensor (ffn_gate_up_exps) and keeps an
# always-on shared expert (the layer's dense ffn_{gate,up,down}). We interleave one dense
# layer (no ffn_gate_inp) and make one layer full-attention (the rest sliding-window) so
# the fixture covers dense/MoE interleaving and the mixed SWA KV cache. Only the two
# expert weight tensors stream; the shared expert, router and gate_inp.scale stay resident.
DENSE_LAYER = 0 # a dense (non-MoE) layer, to exercise interleaving
FULL_ATTN_LAYER = 2 # the one non-SWA layer (rest are sliding-window)
def build_gemma4(out):
tokens, scores, toktypes = build_vocab()
n_vocab = len(tokens)
w = gguf.GGUFWriter(out, "gemma4")
w.add_name("tiny-moe")
w.add_context_length(N_CTX)
w.add_embedding_length(N_EMBD)
w.add_block_count(N_LAYER)
w.add_feed_forward_length(N_FF)
w.add_head_count(N_HEAD)
w.add_head_count_kv(N_HEAD_KV)
w.add_key_length(N_EMBD_HEAD)
w.add_value_length(N_EMBD_HEAD)
w.add_rope_freq_base(ROPE_BASE)
w.add_layer_norm_rms_eps(RMS_EPS)
w.add_expert_count(N_EXPERT)
w.add_expert_used_count(N_EXPERT_USED)
w.add_expert_feed_forward_length(N_FF_EXP)
w.add_file_type(gguf.LlamaFileType.ALL_F32)
# gemma4-specific hparams. One full-attention layer, the rest sliding-window; SWA head
# dims equal the global ones so every layer shares the same shape. Per-layer input
# embeddings are disabled (length 0) to keep the tensor set minimal.
swa_pattern = [i != FULL_ATTN_LAYER for i in range(N_LAYER)]
w.add_sliding_window_pattern(swa_pattern)
w.add_sliding_window(N_CTX)
w.add_key_length_swa(N_EMBD_HEAD)
w.add_value_length_swa(N_EMBD_HEAD)
w.add_embedding_length_per_layer_input(0)
add_tokenizer(w, tokens, scores, toktypes)
# Tied output (no output.weight → llama.cpp reuses token_embd). One shared rope_freqs
# tensor covers the full-attention layer.
w.add_tensor("token_embd.weight", rnd(n_vocab, N_EMBD, seed=1))
w.add_tensor("output_norm.weight", rnd(N_EMBD, seed=2))
w.add_tensor("rope_freqs.weight", rnd(N_EMBD_HEAD // 2, seed=3))
s = 100
for i in range(N_LAYER):
p = f"blk.{i}."
w.add_tensor(p + "attn_norm.weight", rnd(N_EMBD, seed=s + 0))
add_attn_tensors(w, p, s)
w.add_tensor(p + "post_attention_norm.weight", rnd(N_EMBD, seed=s + 7))
# shared / dense FFN (also the shared expert on MoE layers)
w.add_tensor(p + "ffn_norm.weight", rnd(N_EMBD, seed=s + 8))
w.add_tensor(p + "ffn_gate.weight", rnd(N_FF, N_EMBD, seed=s + 9))
w.add_tensor(p + "ffn_up.weight", rnd(N_FF, N_EMBD, seed=s + 10))
w.add_tensor(p + "ffn_down.weight", rnd(N_EMBD, N_FF, seed=s + 11))
w.add_tensor(p + "post_ffw_norm.weight", rnd(N_EMBD, seed=s + 12))
if i != DENSE_LAYER:
# MoE layer: router (+ its required scale), extra norms, and the two streamed
# expert tensors. ffn_gate_up_exps fuses gate+up: dim-1 is 2*N_FF_EXP.
w.add_tensor(p + "ffn_gate_inp.weight", rnd(N_EXPERT, N_EMBD, seed=s + 13))
w.add_tensor(p + "ffn_gate_inp.scale", rnd(N_EMBD, seed=s + 14))
w.add_tensor(p + "pre_ffw_norm_2.weight", rnd(N_EMBD, seed=s + 15))
w.add_tensor(p + "post_ffw_norm_1.weight", rnd(N_EMBD, seed=s + 16))
w.add_tensor(p + "post_ffw_norm_2.weight", rnd(N_EMBD, seed=s + 17))
# experts: dim-2 (numpy axis 0) indexes the expert
w.add_tensor(p + "ffn_gate_up_exps.weight", rnd(N_EXPERT, 2 * N_FF_EXP, N_EMBD, seed=s + 18))
w.add_tensor(p + "ffn_down_exps.weight", rnd(N_EXPERT, N_EMBD, N_FF_EXP, seed=s + 19))
s += 100
w.write_header_to_file()
w.write_kv_data_to_file()
w.write_tensors_to_file()
w.close()
n_moe = N_LAYER - 1
print(f"wrote {out}: gemma4, {N_LAYER} layers ({n_moe} MoE, fused gate_up), "
f"{N_EXPERT} experts (top-{N_EXPERT_USED}), vocab {n_vocab}")
# --- nemotron_h_moe: gate-less expert layout ---------------------------------------
# Nemotron-H MoE (Nemotron 3 / 3.5, e.g. 30B-A3B) has no gate projection: each expert is
# up -> ReLU^2 -> down, so a layer names two expert tensors (ffn_up_exps / ffn_down_exps).
# The stack is hybrid and the block kind is read from per-layer hparams: head_count_kv == 0
# and feed_forward_length == 0 is a Mamba2 block, head_count_kv > 0 an attention block,
# feed_forward_length > 0 a MoE block; no two MoE blocks are adjacent, as in the released
# models. The experts may run in a latent space (ffn_latent_down / ffn_latent_up project in
# and out of it; optional, the 30B has none, the fixture has it to cover the narrower rows),
# the router adds a per-expert bias before a sigmoid top-k, and an always-on shared expert
# (ffn_{up,down}_shexp) sits beside them. A trailing NextN/MTP block names expert tensors
# too but is skipped at load (load_mtp is off), so the fixture also proves an unloaded
# expert-named block never binds.
NEMO_LAYERS = ["mamba", "moe", "attn", "moe", "mamba", "moe"]
NEMO_LATENT = 64 # moe_latent_size: the experts' input/output width
# The latent width makes each expert small, so its hidden width is raised to keep the bank
# (3 layers x 8 experts x 256 KiB) well above the gates' 2 MiB cache, which has to evict.
NEMO_FF_EXP = 512
NEMO_FF_SHEXP = 128
SSM_D_CONV = 4
SSM_D_INNER = 128
SSM_D_STATE = 16
SSM_N_GROUP = 2
SSM_N_HEAD = 4 # ssm time_step_rank: d_inner must divide by it and by n_group
def add_nemotron_moe_tensors(w, p, s, mtp=False):
"""Router, latent projections, the two streamed expert tensors and the shared expert."""
w.add_tensor(p + "ffn_gate_inp.weight", rnd(N_EXPERT, N_EMBD, seed=s + 20))
# Zero selection bias. The loader requires the tensor, but at this model's scale a random
# one would outweigh the logits and pin every token to the same experts, leaving the cache
# and substitution gates nothing to exercise.
w.add_tensor(p + "exp_probs_b.bias", np.zeros(N_EXPERT, dtype=np.float32))
# The loader sizes every expert tensor at the latent width, the MTP block's included,
# but only the trunk's MoE blocks carry the latent projections.
if not mtp:
w.add_tensor(p + "ffn_latent_down.weight", rnd(NEMO_LATENT, N_EMBD, seed=s + 22))
w.add_tensor(p + "ffn_latent_up.weight", rnd(N_EMBD, NEMO_LATENT, seed=s + 23))
# experts: dim-2 (numpy axis 0) indexes the expert
w.add_tensor(p + "ffn_up_exps.weight", rnd(N_EXPERT, NEMO_FF_EXP, NEMO_LATENT, seed=s + 24))
w.add_tensor(p + "ffn_down_exps.weight", rnd(N_EXPERT, NEMO_LATENT, NEMO_FF_EXP, seed=s + 25))
w.add_tensor(p + "ffn_up_shexp.weight", rnd(NEMO_FF_SHEXP, N_EMBD, seed=s + 26))
w.add_tensor(p + "ffn_down_shexp.weight", rnd(N_EMBD, NEMO_FF_SHEXP, seed=s + 27))
def build_nemotron_h_moe(out):
tokens, scores, toktypes = build_vocab()
n_vocab = len(tokens)
n_layer = len(NEMO_LAYERS)
n_all = n_layer + 1 # + the trailing MTP block
kinds = NEMO_LAYERS + ["mtp"]
w = gguf.GGUFWriter(out, "nemotron_h_moe")
w.add_name("tiny-moe")
w.add_context_length(N_CTX)
w.add_embedding_length(N_EMBD)
w.add_block_count(n_all)
w.add_nextn_predict_layers(1)
w.add_feed_forward_length([N_FF if k == "moe" else 0 for k in kinds])
w.add_head_count([N_HEAD] * n_all)
w.add_head_count_kv([N_HEAD_KV if k in ("attn", "mtp") else 0 for k in kinds])
w.add_key_length(N_EMBD_HEAD)
w.add_value_length(N_EMBD_HEAD)
w.add_layer_norm_rms_eps(RMS_EPS)
w.add_layer_norm_eps(RMS_EPS)
w.add_ssm_conv_kernel(SSM_D_CONV)
w.add_ssm_inner_size(SSM_D_INNER)
w.add_ssm_state_size(SSM_D_STATE)
w.add_ssm_group_count(SSM_N_GROUP)
w.add_ssm_time_step_rank(SSM_N_HEAD)
w.add_expert_count(N_EXPERT)
w.add_expert_used_count(N_EXPERT_USED)
w.add_expert_feed_forward_length(NEMO_FF_EXP)
w.add_expert_shared_feed_forward_length(NEMO_FF_SHEXP)
w.add_expert_shared_count(1)
w.add_expert_weights_norm(True)
w.add_expert_weights_scale(1.0)
w.add_moe_latent_size(NEMO_LATENT)
w.add_file_type(gguf.LlamaFileType.ALL_F32)
add_tokenizer(w, tokens, scores, toktypes)
w.add_tensor("token_embd.weight", rnd(n_vocab, N_EMBD, seed=1))
w.add_tensor("output_norm.weight", rnd(N_EMBD, seed=2))
w.add_tensor("output.weight", rnd(n_vocab, N_EMBD, seed=3))
d_xbc = SSM_D_INNER + 2 * SSM_N_GROUP * SSM_D_STATE
d_in_proj = 2 * SSM_D_INNER + 2 * SSM_N_GROUP * SSM_D_STATE + SSM_N_HEAD
s = 100
for i, kind in enumerate(NEMO_LAYERS):
p = f"blk.{i}."
w.add_tensor(p + "attn_norm.weight", rnd(N_EMBD, seed=s + 0))
if kind == "mamba":
w.add_tensor(p + "ssm_in.weight", rnd(d_in_proj, N_EMBD, seed=s + 1))
w.add_tensor(p + "ssm_conv1d.weight", rnd(d_xbc, SSM_D_CONV, seed=s + 2))
w.add_tensor(p + "ssm_conv1d.bias", rnd(d_xbc, seed=s + 3))
w.add_tensor(p + "ssm_dt.bias", rnd(SSM_N_HEAD, seed=s + 4))
# A = -exp(A_log) is negative in a real checkpoint; keep it so the scan decays
w.add_tensor(p + "ssm_a", -np.abs(rnd(SSM_N_HEAD, 1, seed=s + 5)) - 0.5)
w.add_tensor(p + "ssm_d", rnd(SSM_N_HEAD, 1, seed=s + 6))
w.add_tensor(p + "ssm_norm.weight", rnd(SSM_N_GROUP, SSM_D_INNER // SSM_N_GROUP, seed=s + 7))
w.add_tensor(p + "ssm_out.weight", rnd(N_EMBD, SSM_D_INNER, seed=s + 8))
elif kind == "attn":
w.add_tensor(p + "attn_q.weight", rnd(N_EMBD_HEAD * N_HEAD, N_EMBD, seed=s + 1))
w.add_tensor(p + "attn_k.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 2))
w.add_tensor(p + "attn_v.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 3))
w.add_tensor(p + "attn_output.weight", rnd(N_EMBD, N_EMBD_HEAD * N_HEAD, seed=s + 4))
else:
add_nemotron_moe_tensors(w, p, s)
s += 100
# The MTP block folds an attention and a MoE sub-layer into one trailing block.
p = f"blk.{n_layer}."
w.add_tensor(p + "nextn.enorm.weight", rnd(N_EMBD, seed=s + 1))
w.add_tensor(p + "nextn.hnorm.weight", rnd(N_EMBD, seed=s + 2))
w.add_tensor(p + "nextn.eh_proj.weight", rnd(N_EMBD, 2 * N_EMBD, seed=s + 3))
w.add_tensor(p + "nextn.shared_head_norm.weight", rnd(N_EMBD, seed=s + 4))
w.add_tensor(p + "attn_norm.weight", rnd(N_EMBD, seed=s + 5))
w.add_tensor(p + "attn_q.weight", rnd(N_EMBD_HEAD * N_HEAD, N_EMBD, seed=s + 6))
w.add_tensor(p + "attn_k.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 7))
w.add_tensor(p + "attn_v.weight", rnd(N_EMBD_GQA, N_EMBD, seed=s + 8))
w.add_tensor(p + "attn_output.weight", rnd(N_EMBD, N_EMBD_HEAD * N_HEAD, seed=s + 9))
w.add_tensor(p + "post_attention_norm.weight", rnd(N_EMBD, seed=s + 10))
add_nemotron_moe_tensors(w, p, s, mtp=True)
w.write_header_to_file()
w.write_kv_data_to_file()
w.write_tensors_to_file()
w.close()
n_moe = NEMO_LAYERS.count("moe")
print(f"wrote {out}: nemotron_h_moe, {n_layer} layers ({n_moe} MoE, gate-less) + 1 MTP, "
f"{N_EXPERT} experts (top-{N_EXPERT_USED}), vocab {n_vocab}")
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--arch", choices=["qwen3moe", "gemma4", "nemotron_h_moe"], default="qwen3moe")
ap.add_argument("--out", default="tiny-moe.gguf")
ap.add_argument("--split-max-tensors", type=int, default=0,
help="emit a sharded gguf (N tensors per shard, metadata-only first shard)")
args = ap.parse_args()
if args.arch != "qwen3moe" and args.split_max_tensors:
raise SystemExit("--split-max-tensors is exercised via the qwen3moe fixture only")
if args.arch == "gemma4":
build_gemma4(args.out)
elif args.arch == "nemotron_h_moe":
build_nemotron_h_moe(args.out)
else:
build_qwen3moe(args.out, args.split_max_tensors)
if __name__ == "__main__":
main()