Revert "feat(moe): --drop-cold-experts — spend quality only where it buys I/O (#94)"

This reverts 45a90a2.

The feature merged before the evidence for its shipping default did. The
replay numbers argue the shape of the trade is favourable, but no on-device
A/B is published in this repository, and the app default it landed with
(75%) changes model output for every user of the demo app -- and changes it
non-reproducibly, which no other setting in this engine does.

Nothing was found wrong with the code. This is a sequencing decision: the
work returns as a pull request, with the app default back to off, so the
measurement lands before the default does.

Reverted rather than force-pushed: main is public and this commit was
already pushed, so the history stays honest about what happened.
This commit is contained in:
Helldez 2026-07-22 15:47:14 +02:00
parent 45a90a2df5
commit 79611654cd
31 changed files with 39 additions and 981 deletions

View file

@ -34,10 +34,6 @@ data class AppSettings(
val overlap: Boolean = true, // read the next experts while the current layer computes
val denseWeights: DenseWeights = DenseWeights.ANON, // dense (non-expert) weight residency policy
val prefetchLayers: Int = 0, // temporal prefetch depth K (0 = off); needs the cache
// Cache-aware expert dropping, as a PERCENTAGE of the uniform share 1/top-k (0 = off, 100 = the
// share itself). Stored as an Int because the settings are integer rungs; the flag takes a
// fraction. LOSSY and cache-dependent — it changes the output, and not reproducibly.
val dropColdPct: Int = 75,
val thinking: Boolean = false, // reasoning; off passes --no-think (enable_thinking=false)
val metricsCsv: Boolean = true, // write the engine's per-token CSV for this session (--csv)
) {
@ -93,11 +89,6 @@ data class AppSettings(
// Auto sizing is a live LRU cache, so it satisfies the prefetch cache requirement.
val cacheOn = cacheMb == CACHE_AUTO || cacheMb > 0
if (prefetchLayers > 0 && cacheOn) a += listOf("--prefetch", prefetchLayers.toString())
// Cache-aware dropping needs a live cache to ask about residency — with the cache off
// every expert reads as a miss and the engine rejects the combination outright, so the
// same cacheOn condition that guards prefetch guards this. The engine takes a fraction
// of the uniform share; the setting is stored as a percentage.
if (dropColdPct > 0 && cacheOn) a += listOf("--drop-cold-experts", (dropColdPct / 100.0).toString())
}
return a
}
@ -110,7 +101,7 @@ data class AppSettings(
*/
fun sessionSignature(modelPath: String): String =
listOf(modelPath, mmap, cacheMb, cacheCeilMb, ioThreads, threads, nExpertUsed, oDirect,
overlap, denseWeights, prefetchLayers, dropColdPct)
overlap, denseWeights, prefetchLayers)
.joinToString("|")
fun save(ctx: Context) {
@ -123,7 +114,6 @@ data class AppSettings(
.putBoolean("overlap", overlap)
.putString("denseWeights", denseWeights.name)
.putInt("prefetchLayers", prefetchLayers)
.putInt("dropColdPct", dropColdPct)
.putBoolean("thinking", thinking)
.putBoolean("metricsCsv", metricsCsv)
.apply()
@ -188,10 +178,6 @@ data class AppSettings(
// 0 = model default (top-k as trained). 6/4/3/2 trade output quality for tok/s (fewer routed experts).
val N_EXPERT_CHOICES = intArrayOf(0, 6, 4, 3, 2)
val PREFETCH_CHOICES = intArrayOf(0, 1, 2, 4)
// Percent of the uniform share 1/top-k. 100 is the share itself and the useful maximum:
// above it the threshold could exceed every weight in a routing. The rungs below it are the
// conservative half of the curve, where the replay already beats a top-k cut on both axes.
val DROP_COLD_CHOICES = intArrayOf(0, 50, 75, 100)
val THREAD_CHOICES = intArrayOf(2, 4, 6, 8)
val NPREDICT_CHOICES = intArrayOf(16, 32, 48, 64, 128, 256, 512, 1024, 2048)
@ -220,7 +206,6 @@ data class AppSettings(
}
},
prefetchLayers = p.getInt("prefetchLayers", d.prefetchLayers),
dropColdPct = p.getInt("dropColdPct", d.dropColdPct),
thinking = p.getBoolean("thinking", d.thinking),
metricsCsv = p.getBoolean("metricsCsv", d.metricsCsv),
)

View file

@ -144,25 +144,6 @@ fun SettingsScreen(current: AppSettings, onChange: (AppSettings) -> Unit, onBack
"but the output changes — a speed/quality trade-off.",
fontSize = 12.sp, color = MaterialTheme.colorScheme.onSurfaceVariant,
)
IntSetting(
"Drop cold experts (% of even share)", AppSettings.DROP_COLD_CHOICES, current.dropColdPct,
format = { if (it == 0) "off" else "$it%" },
// Unlike top-k, this one asks the expert source what is resident, so it needs
// both the streamer and a live cache — the same condition prefetch is under.
enabled = !current.mmap &&
(current.cacheMb == AppSettings.CACHE_AUTO || current.cacheMb > 0),
) { onChange(current.copy(dropColdPct = it)) }
Text(
"Experimental. What slows a token down is reading an expert that is not already in RAM. " +
"This skips such an expert when the router barely wanted it anyway — below the chosen " +
"share of an even split. With 8 active experts an even split is 12.5% each, so 75% " +
"means \"skip it if it carries less than 9.4% of the routing\".\n\n" +
"Experts already in RAM always run, however small their weight: they cost no read. The " +
"strongest expert of each routing is never skipped.\n\n" +
"Higher is faster and rougher. Like Active experts, the reply changes — but unlike it, " +
"not the same way twice: what gets skipped depends on what the cache happened to hold.",
fontSize = 12.sp, color = MaterialTheme.colorScheme.onSurfaceVariant,
)
}
Section("Compute") {