diff --git a/CHANGELOG.md b/CHANGELOG.md index f99edf7..470f6fd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -77,6 +77,17 @@ Semantic Versioning. Listed under the app's Experimental group. ### Changed +- **Every run says which mode it ran in, and `--moe-stream` now brings a cache with it.** A first + command with nothing but `-m`, `-p` and `-t` ran plain llama.cpp on mmap: no streaming, no cache, + and a dense policy that only applies once streaming is on. The report said none of this, because + the `moe-stream:` block only prints when streaming is enabled, so a baseline run read as a + measurement of this engine. Two changes: a `mode:` line is printed unconditionally, naming the + flag when a MoE model ran without streaming, and `--cache-mb` defaults to `auto` whenever + `--moe-stream` is on, since the previous default of 0 meant the cache was off. On a 16 GB host + streaming Qwen3.6-35B-A3B Q4_K_M, the cache default alone is 1.16 to 2.35 tok/s and 585 to 238 + MiB read per token, same output. An explicit `--cache-mb` or `BMOE_CACHE_MB` still wins, + including an explicit 0, and the library's own default is unchanged: the CLI resolves this, so an + embedder passing 0 still means no cache. Reported by @eiffel31 (#186). - The CSV summary trailer gains `row_table_MiB`, `row_resident_MiB`, `row_rows`, `row_reads`, `row_read_MiB`, `row_evictions` and `row_io_errors`, and a `moe-rows:` end-of-run line appears when a table qualified. All are absent from the per-token rows, which the policy does not touch. diff --git a/cli/main.cpp b/cli/main.cpp index c74db02..ed1f4ff 100644 --- a/cli/main.cpp +++ b/cli/main.cpp @@ -441,6 +441,7 @@ static void print_usage(const char * argv0) { " MoE expert streaming:\n" " --moe-stream stream only the routed experts per token (MoE models)\n" " --cache-mb N|auto LRU expert cache budget in MiB (0=off, or >=%d); auto=size to device\n" + " (default: auto whenever --moe-stream is on)\n" " --cache-floor-mb N with --cache-mb auto: RAM to leave free (default 1536)\n" " --cache-ceil-mb N with --cache-mb auto: upper bound on the budget (0 = no cap)\n" " --io-threads N parallel expert-read lanes [1..%d] (default 4)\n" @@ -782,6 +783,13 @@ int main(int argc, char ** argv) { if (!seen.count("--predict-log")) cfg.moe.predict_log = env_int("BMOE_PREDICT_LOG", 0) != 0; if (!seen.count("--predict-prefetch")) cfg.moe.predict_prefetch = env_int("BMOE_PREDICT_PREFETCH", 0) != 0; + // A default the CLI resolves rather than the library, so an embedder's explicit 0 keeps meaning + // "no cache". With streaming on, a budget of 0 re-reads every routed expert from flash every + // token, which is never what someone who just typed --moe-stream wanted (#186). An explicit + // --cache-mb or BMOE_CACHE_MB still wins, including an explicit 0. + if (cfg.moe.enabled && !cfg.moe.cache_auto && !seen.count("--cache-mb") && std::getenv("BMOE_CACHE_MB") == nullptr) + cfg.moe.cache_auto = true; + if (cfg.model_path.empty()) { print_usage(argv[0]); // Double-clicked: without this the window closes before the usage can be read, and the @@ -984,6 +992,35 @@ int main(int argc, char ** argv) { std::printf("prefill: %d tokens, %.3f s (%.1f tok/s) | model load %.3f s | TTFT %.3f s\n", s.n_prompt, s.prefill_seconds, prefill_tps, s.load_seconds, s.load_seconds + s.prefill_seconds); } + // What this run actually was. Printed unconditionally, because its absence was the defect: with + // streaming off the engine is plain llama.cpp on mmap, every line below is silent, and a report + // that only ever describes streaming let a baseline run read as a measurement of this project + // (#186). Naming the flag here is cheaper than a doc nobody reaches from a terminal. + { + const char * dense = cfg.moe.dense_weights == DenseWeightsMode::Mmap ? "mmap" + : cfg.moe.dense_weights == DenseWeightsMode::Warmed ? "warm" + : cfg.moe.dense_weights == DenseWeightsMode::Pinned ? "ahwb" + : "anon"; + if (cfg.moe.enabled) { + char cache[64]; + if (cfg.moe.cache_auto) + std::snprintf(cache, sizeof(cache), "cache auto"); + else if (cfg.moe.cache_mb > 0) + std::snprintf(cache, sizeof(cache), "cache %d MiB", cfg.moe.cache_mb); + else + std::snprintf(cache, sizeof(cache), "cache off"); + std::printf("mode: expert streaming, %s, dense %s%s\n", cache, dense, cfg.moe.overlap ? ", overlap" : ""); + } else if (!s.arch.empty() && find_moe_recipe(s.arch.c_str())) { + std::printf("mode: mmap. Expert streaming is OFF on a MoE model (%s), so this run is a " + "baseline, not this engine: add --moe-stream --overlap to stream the routed " + "experts from flash.\n", + s.arch.c_str()); + } else { + std::printf("mode: mmap (%s is not a MoE architecture this build streams; --list-archs " + "lists the supported ones)\n", + s.arch.empty() ? "the model" : s.arch.c_str()); + } + } if (cfg.moe.enabled) { std::printf("moe-stream: read %.1f MiB (%.2f MiB/token), decode %.3f s/token " "(compute %.3f + cache mgmt %.3f + flash I/O %.3f s/token, %.0f MiB/s)\n", diff --git a/core/include/bmoe/metrics.h b/core/include/bmoe/metrics.h index c382d1d..7b83edf 100644 --- a/core/include/bmoe/metrics.h +++ b/core/include/bmoe/metrics.h @@ -99,6 +99,11 @@ struct TokenMetrics { }; struct RunSummary { + // The loaded model's architecture, as gguf reports it. Carried here so a caller can tell what + // the run was capable of and not only what it did: a MoE arch that ran without streaming is the + // baseline, not a measurement of this engine (see the CLI's mode line). + std::string arch; + int n_generated = 0; double gen_seconds = 0.0; double s_per_token = 0.0; diff --git a/core/src/engine/session.cpp b/core/src/engine/session.cpp index aa55674..04c4b6f 100644 --- a/core/src/engine/session.cpp +++ b/core/src/engine/session.cpp @@ -1675,6 +1675,7 @@ RunResult Session::generate(const GenerateRequest & req, // ── summary ── RunSummary & s = res.summary; + s.arch = im.arch; s.n_generated = n_gen; s.gen_seconds = gen_seconds; s.s_per_token = n_gen ? gen_seconds / n_gen : 0.0; diff --git a/docs/cache-sizing.md b/docs/cache-sizing.md index b342922..c95a539 100644 --- a/docs/cache-sizing.md +++ b/docs/cache-sizing.md @@ -123,7 +123,7 @@ see the ordering warning below. | Flag | Meaning | |---|---| -| `--cache-mb auto` | size the cache to the device instead of a fixed MiB (mutually exclusive with a numeric `--cache-mb`) | +| `--cache-mb auto` | size the cache to the device instead of a fixed MiB (mutually exclusive with a numeric `--cache-mb`). **The CLI's default whenever `--moe-stream` is on**: pass a number, or `0` for no cache, to override it. The library's own default is still no cache, so an embedder passing 0 keeps meaning it | | `--cache-floor-mb N` | RAM to leave free for the rest of the system when auto-sizing (default 1536) | | `--cache-ceil-mb N` | upper bound on the auto-sized budget (0 = no cap). Use it — uncapped `auto` over-asks | | `--dense-weights mmap\|warm\|anon` | the dense (non-expert) weight policy. `warm` is the load-time page-cache sweep described above; `mmap` skips it; `anon` (default) reads the dense set via O_DIRECT into anonymous buffers instead, which is the right answer well past RAM — see [benchmarks-gpt-oss.md](benchmarks-gpt-oss.md). `--no-warm-dense` and `--dense-odirect` are deprecated aliases for `mmap` and `anon` | diff --git a/docs/telemetry.md b/docs/telemetry.md index 30200da..984c89e 100644 --- a/docs/telemetry.md +++ b/docs/telemetry.md @@ -116,10 +116,17 @@ BMOE_PROGRESS {"step":,"steps":,"wall_ms":,"io_ms":, === perf === generation: tokens, s/token ( tok/s) compute: % CPU occupancy ( cpu-s/token over threads), major faults/token +mode: expert streaming, cache MiB|off>, dense [, overlap] moe-stream: read MiB ( MiB/token), decode s/token (compute + cache mgmt + flash I/O s/token, MiB/s) moe-cache: % hit, resident MiB ``` +The `mode:` line is printed on every run, streaming or not, and is the first thing to read: without +`--moe-stream` the engine is plain llama.cpp on mmap and every `moe-*` line below is absent, so a +report missing them is a baseline and not a measurement of this engine. On a MoE architecture the +build has a recipe for, that case names the flag; on any other model it says the architecture is not +one this build streams. + The `compute:` line decomposes the residual: low CPU occupancy points at a throttled/preempted core (a frequency cap, a co-resident process) rather than heavy math, and non-zero major faults/token means dense weights were re-faulting from flash inside the decode. It is omitted on