BigMoeOnEdge/CMakeLists.txt
Helldez 927e2d3b31
feat(moe): Qwen3.8-Flash-Next support (#172)
Qwen3.8-Flash-Next (qwen4exp): 125B total, ~6B active, 512 routed experts at
top-10 plus one shared, 48 hybrid gated-delta SSM / sparse attention layers,
and a 51B n-gram embedding table. One registry row streams the experts; a
dense-policy guard keeps the n-gram table (larger than any phone's RAM)
mmap'd under every mode so pinned and anonymous dense weights survive load.
Runs on the 12 GB test phone at ~2 tok/s with pinned dense weights, compute-
bound, and sits in the app catalog as a three-shard download. Submodule
pinned to upstream master b10666, the first with the architecture merged,
with the expert-ready hook on top. README hero clip, changelog and docs
updated. App 0.22.0 (versionCode 37).
2026-08-28 10:07:43 +02:00

72 lines
3.6 KiB
CMake

# BigMoeOnEdge — top-level build.
#
# The engine (`bmoe_core`) depends only on its own public headers under include/bmoe
# and on llama.cpp's PUBLIC C API (llama.h / ggml.h). The serial streaming path never
# touches llama.cpp internals: MoE expert streaming is driven entirely through the public
# eval-callback and public gguf accessors, so it tracks upstream with nothing more than a
# submodule pointer bump — no in-tree patch. The one exception is the optional `--overlap`
# feature, which needs a per-expert wait point the public API does not expose: the submodule
# pins a fork branch (`bmoe/expert-ready-hook` on Helldez/llama.cpp) adding one ~25-line
# readiness hook, detected here as BMOE_HAVE_EXPERT_READY_HOOK. It is zero-cost when absent,
# the serial path still builds against stock upstream, and it sunsets the moment upstream
# ships an equivalent callback. See docs/seam.md § 3.
#
# llama.cpp is an optional dependency at configure time: when the submodule is absent
# only the pure-policy code (config validation) compiles, so the scaffold and a subset
# of CI stay green before the native dependency is fetched.
cmake_minimum_required(VERSION 3.21)
# The version is declared here and nowhere else in the build: the engine reports it (bmoe-cli
# --version, and the run-parameter preamble of every metrics CSV), so a committed benchmark file
# says which engine produced it. Keep it in step with CHANGELOG.md and the app's versionName.
project(bigmoeonedge VERSION 0.22.0 LANGUAGES CXX)
set(CMAKE_CXX_STANDARD 17)
set(CMAKE_CXX_STANDARD_REQUIRED ON)
set(CMAKE_CXX_EXTENSIONS OFF)
if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES)
set(CMAKE_BUILD_TYPE Release CACHE STRING "" FORCE)
endif()
option(BMOE_BUILD_TESTS "Build byte-identity gates and unit tests" ON)
option(BMOE_BUILD_CLI "Build the bmoe-cli host tool" ON)
# Off by default: diagnostics, not part of the product. They need no llama.cpp, so they can be
# configured and cross-compiled on their own without building the engine.
option(BMOE_BUILD_TOOLS "Build standalone diagnostic tools (bmoe-iobench)" OFF)
# --- Optional native dependency: llama.cpp ----------------------------------------
# The streaming seam uses only the public C API (llama.h). We additionally link llama.cpp's
# `common` layer for one thing: chat-template rendering + reasoning parsing (common_chat_*),
# so prompt formatting and thinking-extraction are template-driven per model instead of
# hardcoded. Unlike the streaming seam, `common` is NOT a stable API — it can change between
# upstream versions, so a submodule bump may need the chat glue in runtime.cpp updated (the
# gates catch it at build time). That trade-off is deliberate; see docs/seam.md.
set(BMOE_HAVE_LLAMA OFF)
if(EXISTS "${CMAKE_CURRENT_SOURCE_DIR}/third_party/llama.cpp/CMakeLists.txt")
set(BMOE_HAVE_LLAMA ON)
# Upstream examples/tests/server stay off; build the common utils we link for chat.
set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_SERVER OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_COMMON ON CACHE BOOL "" FORCE)
set(LLAMA_CURL OFF CACHE BOOL "" FORCE) # common's model-download path unused
add_subdirectory(third_party/llama.cpp EXCLUDE_FROM_ALL)
endif()
add_subdirectory(core)
if(BMOE_BUILD_CLI AND BMOE_HAVE_LLAMA)
add_subdirectory(cli)
endif()
if(BMOE_BUILD_TESTS AND BMOE_HAVE_LLAMA)
enable_testing()
add_subdirectory(tests)
endif()
if(BMOE_BUILD_TOOLS)
add_subdirectory(tools)
endif()
message(STATUS "BigMoeOnEdge: llama.cpp present = ${BMOE_HAVE_LLAMA}")