mirror of
https://github.com/Helldez/BigMoeOnEdge.git
synced 2026-10-03 03:25:42 +00:00
feat(engine): --route-ahead N — commit decode routing to the N-layers-early prediction (#142)
* feat(engine): --route-ahead N — commit decode routing to the N-layers-early prediction Every prefetch lives under the same ceiling: layer L's routing needs layer L-1's output, so any predictor working earlier is approximate and every speculated read can miss. This inverts the bet. The expert selection of decode layer L is REPLACED by the ranking layer L's own gate matrix produced on the hidden state N layers back in the same forward pass, so the selection is known N layers early and cannot miss. With the cache on, those reads are issued the moment the selection is fixed. Lossy by construction: it changes the output, and roughly a fifth of slots route to a different expert than the router chose at N=1. Off by default, mutually exclusive with both prefetchers and with the prediction probe, since each would speculate on a future this policy has already decided. Quality is measured rather than assumed: the committed generations match the baseline on a four-prompt objective battery, on a long essay, and on a second model of a different generation and quantization; output stays deterministic across repeated runs, which expert dropping does not. See docs/route-ahead.md. Squashed from the seventeen commits of exp/route-ahead: the branch predated the multi-shard and speculative-decoding work, and replaying it commit by commit meant resolving the same two collisions seventeen times over. The history is preserved on the pull request; what lands here is what a squash-merge would have produced anyway. * fix(engine): refuse route-ahead alongside self-speculation, and say why Running the two together on a real model committed NOTHING: 0 routings taken, 249 passed through. A verify decode is several positions wide and the policy correctly declines each one. But it still charged for itself — the prediction GEMVs ran (2.6 ms/token of worker CPU) and its early reads degraded into ordinary speculation, falling from 100% useful to 81%. Cost with no commitment is worse than either feature alone, and nothing told the user. validate() now rejects the pair the way it already rejects route-ahead beside the two prefetchers, the app stops emitting the flag and greys the row out, and the config test covers both draft sources. Making the combination work is a different change: commit the whole verify batch to one selection. That is written and measured, and it costs draft acceptance (70% to 53%) while route-ahead alone still won, so the exclusion is the honest state today rather than a limitation to be worked around. Found by the desktop smoke run, not by the gates.
This commit is contained in:
parent
49c72e7ce5
commit
4645dc6b09
21 changed files with 1391 additions and 54 deletions
|
|
@ -47,6 +47,13 @@ data class AppSettings(
|
|||
// spends no flash and only protects predicted residents from eviction). Defaults to 0 because
|
||||
// the matched-pair A/B showed read-ahead losing on a saturated flash (docs/expert-prediction.md).
|
||||
val predictSpecMax: Int = 0,
|
||||
// Route-ahead (experimental, LOSSY): decode routing at layer L is COMMITTED to the prediction
|
||||
// made N layers earlier in the same forward pass, and with the cache on the committed experts
|
||||
// are read that early — reads that can never be wasted, since they ARE the routing. Changes
|
||||
// the output (~20% of slots re-route at N=1; quality held in the first host A/B). Excludes
|
||||
// both prefetchers, so sessionArgv only emits it when they are off. 0 = off, the default
|
||||
// until the on-device A/B earns it more.
|
||||
val routeAhead: Int = 0,
|
||||
// Cache-aware expert dropping, as a PERCENTAGE of the uniform share 1/top-k (0 = off, 100 = the
|
||||
// share itself). Stored as an Int because the settings are integer rungs; the flag takes a
|
||||
// fraction. LOSSY and cache-dependent — it changes the output, and not reproducibly.
|
||||
|
|
@ -144,6 +151,14 @@ data class AppSettings(
|
|||
a += "--predict-prefetch"
|
||||
a += listOf("--predict-spec-max", predictSpecMax.toString())
|
||||
}
|
||||
// Route-ahead excludes both prefetchers (the engine refuses the pairs: they would
|
||||
// speculate a future route-ahead has already fixed) and self-speculation (a verify
|
||||
// decode is several positions wide, and route-ahead declines to commit on every one of
|
||||
// them while still paying for the prediction). It works without the cache too — the
|
||||
// routing still commits — but only reads early when the cache is on.
|
||||
if (routeAhead > 0 && prefetchLayers == 0 && !predictPrefetch && spec == SPEC_OFF) {
|
||||
a += listOf("--route-ahead", routeAhead.toString())
|
||||
}
|
||||
// Cache-aware dropping needs a live cache to ask about residency — with the cache off
|
||||
// every expert reads as a miss and the engine rejects the combination outright, so the
|
||||
// same cacheOn condition that guards prefetch guards this. The engine takes a fraction
|
||||
|
|
@ -170,8 +185,8 @@ data class AppSettings(
|
|||
*/
|
||||
fun sessionSignature(modelPath: String): String =
|
||||
listOf(modelPath, mmap, cacheMb, cacheCeilMb, ioThreads, threads, nExpertUsed, sessionCtx, oDirect,
|
||||
overlap, denseWeights, prefetchLayers, predictPrefetch, predictSpecMax, dropColdPct,
|
||||
spec, mtpDraft, mtpPMinPct)
|
||||
overlap, denseWeights, prefetchLayers, predictPrefetch, predictSpecMax, routeAhead,
|
||||
dropColdPct, spec, mtpDraft, mtpPMinPct)
|
||||
.joinToString("|")
|
||||
|
||||
fun save(ctx: Context) {
|
||||
|
|
@ -186,6 +201,7 @@ data class AppSettings(
|
|||
.putInt("prefetchLayers", prefetchLayers)
|
||||
.putBoolean("predictPrefetch", predictPrefetch)
|
||||
.putInt("predictSpecMax", predictSpecMax)
|
||||
.putInt("routeAhead", routeAhead)
|
||||
.putInt("dropColdPct", dropColdPct)
|
||||
.putInt("sessionCtx", sessionCtx)
|
||||
.putString("spec", spec).putInt("mtpDraft", mtpDraft).putInt("mtpPMinPct", mtpPMinPct)
|
||||
|
|
@ -287,6 +303,10 @@ data class AppSettings(
|
|||
// default — the matched-pair A/B showed 2 losing −21% on a saturated flash; anything above
|
||||
// it re-buys the measured full-speculation pathology.
|
||||
val PREDICT_SPEC_CHOICES = intArrayOf(0, 1, 2, 4)
|
||||
// Route-ahead depth: how many layers early the routing is committed (and read). Each layer
|
||||
// of depth widens the I/O window and the routing perturbation together; 1 is the measured
|
||||
// sweet spot on the host, 4 the edge where damage first showed.
|
||||
val ROUTE_AHEAD_CHOICES = intArrayOf(0, 1, 2, 4)
|
||||
// Percent of the uniform share 1/top-k. 100 is the share itself and the useful maximum:
|
||||
// above it the threshold could exceed every weight in a routing. The rungs below it are the
|
||||
// conservative half of the curve, where the replay already beats a top-k cut on both axes.
|
||||
|
|
@ -321,6 +341,7 @@ data class AppSettings(
|
|||
prefetchLayers = p.getInt("prefetchLayers", d.prefetchLayers),
|
||||
predictPrefetch = p.getBoolean("predictPrefetch", d.predictPrefetch),
|
||||
predictSpecMax = p.getInt("predictSpecMax", d.predictSpecMax),
|
||||
routeAhead = p.getInt("routeAhead", d.routeAhead),
|
||||
dropColdPct = p.getInt("dropColdPct", d.dropColdPct),
|
||||
sessionCtx = p.getInt("sessionCtx", d.sessionCtx),
|
||||
spec = run {
|
||||
|
|
|
|||
|
|
@ -44,6 +44,11 @@ object MetricFields {
|
|||
|
||||
MetricField("loop_overhead_ms", "time between tokens", "everything outside llama_decode: sampling, detokenization, rendering the answer, writing the sinks. It falls outside wall_ms and so outside the reported tok/s, which is why it needs a column — work that moves only this number is paid on every token and shows up nowhere else. On the first token, the gap from the end of prefill to the first decode", Better.LOWER),
|
||||
|
||||
MetricField("drain_ms", "wait on previous batch", "overlap only: eval-thread time waiting for the PREVIOUS layer's reads to finish before reusing their slots — part of compute_ms, named so a large compute residual can be attributed instead of guessed at", Better.LOWER),
|
||||
MetricField("adopt_ms", "wait adopting committed reads", "route-ahead only: the load waiting for its own committed speculative reads to complete before staging — part of mgmt_ms", Better.LOWER),
|
||||
MetricField("ra_issue_ms", "issuing early reads", "route-ahead only: eval-thread time issuing the committed selection's reads (settle, residency, retain, prefetch) — part of compute_ms", Better.LOWER),
|
||||
MetricField("ra_wd_ms", "watchdog control", "route-ahead only: the sampled fresh-gate control's exact GEMV on the eval thread — part of compute_ms", Better.LOWER),
|
||||
|
||||
MetricField("mem_available_mib", "device 'available' (it lies)", "what the kernel claims is free — it counts our own mmap'd weights as reclaimable, so it over-states headroom", Better.NEUTRAL),
|
||||
MetricField("mem_free_mib", "device free", "truly free RAM, before any reclaim", Better.NEUTRAL),
|
||||
MetricField("swap_free_mib", "swap free", "zram space left before the device is truly out of room", Better.NEUTRAL),
|
||||
|
|
@ -89,6 +94,7 @@ object ConfigFields {
|
|||
ConfigField("io_two_wave", "the first projection's reads were published to the lanes as soon as they were staged, rather than after the whole layer"),
|
||||
ConfigField("load_all", "every expert was loaded each token, routing ignored. An A/B baseline"),
|
||||
ConfigField("prefetch", "temporal prefetch depth in layers, betting this token routes like the last. 0 = off"),
|
||||
ConfigField("route_ahead", "each layer's expert choice was COMMITTED to the prediction made this many layers earlier in the same forward pass, and read that early. Lossy: a fraction of choices differ from the router's own. 0 = off"),
|
||||
ConfigField("predict_prefetch", "the next layer's router ran early, and the cache acted on that prediction instead of the previous token's routing"),
|
||||
ConfigField("predict_spec_max", "how many predicted, non-resident experts were read ahead. 0 = retention only, which spends no flash and merely protects predicted residents from eviction. Recorded but unused when predict_prefetch is off"),
|
||||
ConfigField("predict_log", "prediction-accuracy probe. A diagnostic, not a shipping setting"),
|
||||
|
|
|
|||
|
|
@ -46,6 +46,7 @@ internal class Csv(
|
|||
// dropped experts is not the same KIND of run as one that did not, and two compare
|
||||
// legends that differ only by this used to read as identical (#136).
|
||||
if (info["predict_prefetch"] == "1") "predict" else null,
|
||||
info["route_ahead"]?.takeIf { it != "0" }?.let { "route-ahead $it" },
|
||||
info["drop_cold_frac"]?.takeIf { (it.toFloatOrNull() ?: 0f) > 0f }?.let { "drop $it" },
|
||||
// Same argument, and stronger: under speculation a decode confirms a whole group, so
|
||||
// the per-token rows are not even accounted the same way (mtp_batch). Two runs that
|
||||
|
|
|
|||
|
|
@ -410,6 +410,7 @@ private val CONFIG_ORDER = listOf(
|
|||
"io_two_wave" to "Two-wave publish",
|
||||
"load_all" to "Load all experts",
|
||||
"prefetch" to "Temporal prefetch",
|
||||
"route_ahead" to "Route-ahead (layers)",
|
||||
"predict_prefetch" to "Predictive prefetch",
|
||||
"predict_spec_max" to "Speculated misses/layer",
|
||||
"predict_log" to "Prediction probe",
|
||||
|
|
@ -461,7 +462,7 @@ private fun prettyConfigValue(key: String, v: String, info: Map<String, String>)
|
|||
// dropping happened" is exactly the misreading this display exists to prevent. The drop
|
||||
// fraction is shown as the engine took it (a fraction of the uniform share 1/top-k, not of the
|
||||
// routing) so it matches --drop-cold-experts and the settings screen.
|
||||
(key == "prefetch" || key == "drop_cold_frac") && v.toFloatOrNull() == 0f -> "off"
|
||||
(key == "prefetch" || key == "route_ahead" || key == "drop_cold_frac") && v.toFloatOrNull() == 0f -> "off"
|
||||
key == "predict_spec_max" && v == "0" -> "0 (retention only)"
|
||||
else -> v
|
||||
}
|
||||
|
|
|
|||
|
|
@ -120,7 +120,7 @@ fun SettingsScreen(current: AppSettings, onChange: (AppSettings) -> Unit, onBack
|
|||
format = { if (it == 0) "off" else "$it" },
|
||||
// Mutually exclusive with predictive prefetch: two predictors would speculate
|
||||
// the same future twice, and the engine refuses the pair.
|
||||
enabled = stream && cacheOn && !current.predictPrefetch,
|
||||
enabled = stream && cacheOn && !current.predictPrefetch && current.routeAhead == 0,
|
||||
) { onChange(current.copy(prefetchLayers = it)) }
|
||||
Text(
|
||||
"Experimental. Bets a layer will reuse the experts it picked for the previous token, and " +
|
||||
|
|
@ -129,10 +129,12 @@ fun SettingsScreen(current: AppSettings, onChange: (AppSettings) -> Unit, onBack
|
|||
)
|
||||
SwitchRow(
|
||||
"Predictive prefetch (experimental)",
|
||||
"Experimental. Rather than betting on the previous token, runs the next layer's own " +
|
||||
"router early to ask which experts it will actually want. Reads ahead within the " +
|
||||
"budget below and protects what it names. Needs the cache; replaces the above.",
|
||||
current.predictPrefetch, enabled = stream && cacheOn && current.prefetchLayers == 0,
|
||||
"Instead of betting on the previous token, asks the next layer's own router one " +
|
||||
"layer early to find out which experts it will actually want. Reads ahead within " +
|
||||
"the budget below and keeps what the prediction names. Needs the cache; replaces " +
|
||||
"temporal prefetch.",
|
||||
current.predictPrefetch,
|
||||
enabled = stream && cacheOn && current.prefetchLayers == 0 && current.routeAhead == 0,
|
||||
) { onChange(current.copy(predictPrefetch = it)) }
|
||||
if (current.predictPrefetch) {
|
||||
IntSetting(
|
||||
|
|
@ -215,6 +217,22 @@ fun SettingsScreen(current: AppSettings, onChange: (AppSettings) -> Unit, onBack
|
|||
"the reading, and changes the reply — a deliberate trade of quality for speed.",
|
||||
fontSize = 12.sp, color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
IntSetting(
|
||||
"Route-ahead (layers)", AppSettings.ROUTE_AHEAD_CHOICES, current.routeAhead,
|
||||
format = { if (it == 0) "off" else "$it" },
|
||||
// Excludes both prefetchers (the engine refuses speculating a future this has
|
||||
// already fixed) and guessing ahead, where a verify pass is several tokens wide
|
||||
// and this declines to commit on all of them while still paying for itself.
|
||||
// Needs streaming; the cache is what turns it into early reads.
|
||||
enabled = !current.mmap && current.prefetchLayers == 0 && !current.predictPrefetch &&
|
||||
current.spec == AppSettings.SPEC_OFF,
|
||||
) { onChange(current.copy(routeAhead = it)) }
|
||||
Text(
|
||||
"Experimental, changes the output. Each layer's expert choice is committed N layers " +
|
||||
"early — so with the cache on their reads start that early and are never wasted. " +
|
||||
"~20% of choices differ from the router's at 1 layer; quality held in the host A/B.",
|
||||
fontSize = 12.sp, color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
IntSetting(
|
||||
"Drop cold experts (% of even share)", AppSettings.DROP_COLD_CHOICES, current.dropColdPct,
|
||||
// The rung labels carry the trade, so the blurb below does not have to repeat
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue