mirror of
https://github.com/Helldez/BigMoeOnEdge.git
synced 2026-10-03 03:25:42 +00:00
feat(moe): serve row-gathered dense tables from flash (--row-stream) (#180)
Dense tables the graph only gathers rows from (the token embedding, on most models) are bound to reserved address space and fetched in 16 KiB slabs inside a bounded LRU window, instead of being read whole and kept resident. Which tables qualify is decided from the captured graph, not from a name list. Byte-identical to the resident reference; -497 MiB pinned on Qwen3.8-Flash-Next and -515 MiB on Qwen3.6-35B on the 12 GB test phone, throughput neutral, off by default. Gates G15a/G15b. App 0.24.0 (unreleased), new page docs/row-gathered-tables.md.
This commit is contained in:
parent
d754dfd6fc
commit
9153dcc27a
26 changed files with 1042 additions and 40 deletions
|
|
@ -119,6 +119,12 @@ with Mmap it is refaulted from flash every token. Its 51B n-gram table is held b
|
|||
stays mmap'd whatever the setting; the engine says so on stderr at load. Keep the expert cache at
|
||||
1000-1500 MiB on a 12 GB phone: the pinned dense set leaves no room for more.
|
||||
|
||||
**Stream row-gathered tables** takes ~500 MiB more off that pinned set on this model, and 515 MiB
|
||||
on Qwen3.6: the token embedding table is read one row per token, so it does not need to be in
|
||||
RAM at all. The reply is identical either way. It is off by default until a long run on a phone
|
||||
says whether the RAM it hands back is worth the reads, which is exactly the kind of thing this
|
||||
app exists to find out.
|
||||
|
||||
## Expected numbers
|
||||
|
||||
On a phone with UFS 4.x storage and ~12 GB RAM, streaming Qwen3-30B-A3B-Q4_K_M with the
|
||||
|
|
@ -150,3 +156,6 @@ Two worth knowing before you turn them on:
|
|||
- **"Decide the experts early"** (`--route-ahead`) commits each layer's routing before that layer
|
||||
runs, so the reads can never be wasted. It changes the reply, and it is refused alongside guessing
|
||||
ahead. See `../../docs/route-ahead.md`.
|
||||
- **"Stream row-gathered tables"** (`--row-stream`) serves the token embedding table out of flash
|
||||
instead of RAM. Lossless, and which tables it applies to is read off the model's own graph, so
|
||||
on a model where none qualify it does nothing. See `../../docs/row-gathered-tables.md`.
|
||||
|
|
|
|||
|
|
@ -51,8 +51,8 @@ android {
|
|||
applicationId 'io.bigmoeonedge.example'
|
||||
minSdk 29
|
||||
targetSdk 34
|
||||
versionCode 38
|
||||
versionName '0.23.0'
|
||||
versionCode 39
|
||||
versionName '0.24.0'
|
||||
buildConfigField 'String', 'GIT_SHA', "\"${gitSha}\""
|
||||
ndk {
|
||||
// The engine ships as prebuilt arm64 binaries staged by build-android.ps1.
|
||||
|
|
|
|||
|
|
@ -58,6 +58,13 @@ data class AppSettings(
|
|||
// share itself). Stored as an Int because the settings are integer rungs; the flag takes a
|
||||
// fraction. LOSSY and cache-dependent — it changes the output, and not reproducibly.
|
||||
val dropColdPct: Int = 75,
|
||||
// Serve the dense tables the graph only GATHERS ROWS from - a token embedding - out of flash
|
||||
// instead of RAM. Which tables qualify is decided by the graph at load, so this is one switch
|
||||
// for every model rather than a per-model list; on a model where nothing qualifies it does
|
||||
// nothing at all. Lossless by construction (the rows read are the rows the graph asks for),
|
||||
// so the only question it raises is whether the reads cost more than the RAM is worth - which
|
||||
// is why it is off until the on-device A/B says otherwise.
|
||||
val rowStream: Boolean = false,
|
||||
// Which source drafts for self-speculation: "off", "mtp" or "ngram". Both verify the same way —
|
||||
// one wider decode, greedy acceptance — and differ only in what a draft costs.
|
||||
//
|
||||
|
|
@ -164,6 +171,10 @@ data class AppSettings(
|
|||
// same cacheOn condition that guards prefetch guards this. The engine takes a fraction
|
||||
// of the uniform share; the setting is stored as a percentage.
|
||||
if (dropColdPct > 0 && cacheOn) a += listOf("--drop-cold-experts", (dropColdPct / 100.0).toString())
|
||||
// Row-gathered dense tables. Inside the streaming block because the tables are
|
||||
// discovered by the streamer's capture pass; independent of the cache and of the
|
||||
// dense-weight mode, since what it changes is which tensors that mode applies to.
|
||||
if (rowStream) a += "--row-stream"
|
||||
}
|
||||
// Outside the streaming block on purpose: speculation is a decode-loop change, not a
|
||||
// residency policy, so it applies to the mmap baseline too — which is what makes an A/B of
|
||||
|
|
@ -209,6 +220,7 @@ data class AppSettings(
|
|||
.putInt("predictSpecMax", predictSpecMax)
|
||||
.putInt("routeAhead", routeAhead)
|
||||
.putInt("dropColdPct", dropColdPct)
|
||||
.putBoolean("rowStream", rowStream)
|
||||
.putInt("sessionCtx", sessionCtx)
|
||||
.putString("spec", spec).putInt("mtpDraft", mtpDraft).putInt("mtpPMinPct", mtpPMinPct)
|
||||
.putBoolean("thinking", thinking)
|
||||
|
|
@ -347,6 +359,7 @@ data class AppSettings(
|
|||
predictSpecMax = p.getInt("predictSpecMax", d.predictSpecMax),
|
||||
routeAhead = p.getInt("routeAhead", d.routeAhead),
|
||||
dropColdPct = p.getInt("dropColdPct", d.dropColdPct),
|
||||
rowStream = p.getBoolean("rowStream", d.rowStream),
|
||||
sessionCtx = p.getInt("sessionCtx", d.sessionCtx),
|
||||
spec = run {
|
||||
val saved = p.getString("spec", null)
|
||||
|
|
|
|||
|
|
@ -117,6 +117,16 @@ fun SettingsScreen(current: AppSettings, onChange: (AppSettings) -> Unit, onBack
|
|||
) { onChange(current.copy(denseWeights = DenseWeights.values()[it])) }
|
||||
Hint(current.denseWeights.blurb)
|
||||
|
||||
SwitchRow(
|
||||
"Stream row-gathered tables",
|
||||
"A dense table the model only reads a few ROWS from per token (the token " +
|
||||
"embedding) does not need to be in RAM: only the rows are read, from flash. " +
|
||||
"Lossless - the output is identical. Which tables qualify is read off the " +
|
||||
"graph at load, so on a model where none do this does nothing.",
|
||||
current.rowStream,
|
||||
enabled = stream,
|
||||
) { onChange(current.copy(rowStream = it)) }
|
||||
|
||||
ExperimentalGroup {
|
||||
IntSetting(
|
||||
"Temporal prefetch (layers)", AppSettings.PREFETCH_CHOICES, current.prefetchLayers,
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue