mirror of
https://github.com/Helldez/BigMoeOnEdge.git
synced 2026-10-03 03:25:42 +00:00
Revert "feat(moe): --drop-cold-experts — spend quality only where it buys I/O (#94)"
This reverts 45a90a2.
The feature merged before the evidence for its shipping default did. The
replay numbers argue the shape of the trade is favourable, but no on-device
A/B is published in this repository, and the app default it landed with
(75%) changes model output for every user of the demo app -- and changes it
non-reproducibly, which no other setting in this engine does.
Nothing was found wrong with the code. This is a sequencing decision: the
work returns as a pull request, with the app default back to off, so the
measurement lands before the default does.
Reverted rather than force-pushed: main is public and this commit was
already pushed, so the history stays honest about what happened.
This commit is contained in:
parent
45a90a2df5
commit
79611654cd
31 changed files with 39 additions and 981 deletions
|
|
@ -34,10 +34,6 @@ data class AppSettings(
|
|||
val overlap: Boolean = true, // read the next experts while the current layer computes
|
||||
val denseWeights: DenseWeights = DenseWeights.ANON, // dense (non-expert) weight residency policy
|
||||
val prefetchLayers: Int = 0, // temporal prefetch depth K (0 = off); needs the cache
|
||||
// Cache-aware expert dropping, as a PERCENTAGE of the uniform share 1/top-k (0 = off, 100 = the
|
||||
// share itself). Stored as an Int because the settings are integer rungs; the flag takes a
|
||||
// fraction. LOSSY and cache-dependent — it changes the output, and not reproducibly.
|
||||
val dropColdPct: Int = 75,
|
||||
val thinking: Boolean = false, // reasoning; off passes --no-think (enable_thinking=false)
|
||||
val metricsCsv: Boolean = true, // write the engine's per-token CSV for this session (--csv)
|
||||
) {
|
||||
|
|
@ -93,11 +89,6 @@ data class AppSettings(
|
|||
// Auto sizing is a live LRU cache, so it satisfies the prefetch cache requirement.
|
||||
val cacheOn = cacheMb == CACHE_AUTO || cacheMb > 0
|
||||
if (prefetchLayers > 0 && cacheOn) a += listOf("--prefetch", prefetchLayers.toString())
|
||||
// Cache-aware dropping needs a live cache to ask about residency — with the cache off
|
||||
// every expert reads as a miss and the engine rejects the combination outright, so the
|
||||
// same cacheOn condition that guards prefetch guards this. The engine takes a fraction
|
||||
// of the uniform share; the setting is stored as a percentage.
|
||||
if (dropColdPct > 0 && cacheOn) a += listOf("--drop-cold-experts", (dropColdPct / 100.0).toString())
|
||||
}
|
||||
return a
|
||||
}
|
||||
|
|
@ -110,7 +101,7 @@ data class AppSettings(
|
|||
*/
|
||||
fun sessionSignature(modelPath: String): String =
|
||||
listOf(modelPath, mmap, cacheMb, cacheCeilMb, ioThreads, threads, nExpertUsed, oDirect,
|
||||
overlap, denseWeights, prefetchLayers, dropColdPct)
|
||||
overlap, denseWeights, prefetchLayers)
|
||||
.joinToString("|")
|
||||
|
||||
fun save(ctx: Context) {
|
||||
|
|
@ -123,7 +114,6 @@ data class AppSettings(
|
|||
.putBoolean("overlap", overlap)
|
||||
.putString("denseWeights", denseWeights.name)
|
||||
.putInt("prefetchLayers", prefetchLayers)
|
||||
.putInt("dropColdPct", dropColdPct)
|
||||
.putBoolean("thinking", thinking)
|
||||
.putBoolean("metricsCsv", metricsCsv)
|
||||
.apply()
|
||||
|
|
@ -188,10 +178,6 @@ data class AppSettings(
|
|||
// 0 = model default (top-k as trained). 6/4/3/2 trade output quality for tok/s (fewer routed experts).
|
||||
val N_EXPERT_CHOICES = intArrayOf(0, 6, 4, 3, 2)
|
||||
val PREFETCH_CHOICES = intArrayOf(0, 1, 2, 4)
|
||||
// Percent of the uniform share 1/top-k. 100 is the share itself and the useful maximum:
|
||||
// above it the threshold could exceed every weight in a routing. The rungs below it are the
|
||||
// conservative half of the curve, where the replay already beats a top-k cut on both axes.
|
||||
val DROP_COLD_CHOICES = intArrayOf(0, 50, 75, 100)
|
||||
val THREAD_CHOICES = intArrayOf(2, 4, 6, 8)
|
||||
val NPREDICT_CHOICES = intArrayOf(16, 32, 48, 64, 128, 256, 512, 1024, 2048)
|
||||
|
||||
|
|
@ -220,7 +206,6 @@ data class AppSettings(
|
|||
}
|
||||
},
|
||||
prefetchLayers = p.getInt("prefetchLayers", d.prefetchLayers),
|
||||
dropColdPct = p.getInt("dropColdPct", d.dropColdPct),
|
||||
thinking = p.getBoolean("thinking", d.thinking),
|
||||
metricsCsv = p.getBoolean("metricsCsv", d.metricsCsv),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -144,25 +144,6 @@ fun SettingsScreen(current: AppSettings, onChange: (AppSettings) -> Unit, onBack
|
|||
"but the output changes — a speed/quality trade-off.",
|
||||
fontSize = 12.sp, color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
IntSetting(
|
||||
"Drop cold experts (% of even share)", AppSettings.DROP_COLD_CHOICES, current.dropColdPct,
|
||||
format = { if (it == 0) "off" else "$it%" },
|
||||
// Unlike top-k, this one asks the expert source what is resident, so it needs
|
||||
// both the streamer and a live cache — the same condition prefetch is under.
|
||||
enabled = !current.mmap &&
|
||||
(current.cacheMb == AppSettings.CACHE_AUTO || current.cacheMb > 0),
|
||||
) { onChange(current.copy(dropColdPct = it)) }
|
||||
Text(
|
||||
"Experimental. What slows a token down is reading an expert that is not already in RAM. " +
|
||||
"This skips such an expert when the router barely wanted it anyway — below the chosen " +
|
||||
"share of an even split. With 8 active experts an even split is 12.5% each, so 75% " +
|
||||
"means \"skip it if it carries less than 9.4% of the routing\".\n\n" +
|
||||
"Experts already in RAM always run, however small their weight: they cost no read. The " +
|
||||
"strongest expert of each routing is never skipped.\n\n" +
|
||||
"Higher is faster and rougher. Like Active experts, the reply changes — but unlike it, " +
|
||||
"not the same way twice: what gets skipped depends on what the cache happened to hold.",
|
||||
fontSize = 12.sp, color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
}
|
||||
|
||||
Section("Compute") {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue