feat(app): a Zero-copy reads switch, so the flag can be A/B'd on the phone

The engine flag is useless in the app without a way to turn it on, and the
device A/B this change still owes has to be run through the app as well as the
CLI. The switch sits under Direct I/O and greys out without it, since O_DIRECT
is what forces the bounce copy in the first place.

Worded as a pure speed knob, because it is one: unlike expert dropping or
route-ahead it is lossless by construction — the same bytes land in the same
places, and the generated text is byte-identical with it on and off.

It joins the session signature, so flipping it reopens the session rather than
silently carrying the old setting into the next generation.
This commit is contained in:
Helldez 2026-08-01 12:32:40 +02:00
parent 447a0f82f3
commit 562b65fd0b
2 changed files with 16 additions and 1 deletions

View file

@ -35,6 +35,11 @@ data class AppSettings(
// shorter context hands the difference back to the expert cache and the dense weights.
val sessionCtx: Int = SESSION_CTX,
val oDirect: Boolean = true, // bypass the page cache
// Read expert slices straight into the cache instead of through a per-lane bounce buffer.
// Needs O_DIRECT (it is what forces the copy in the first place). Lossless by construction —
// the same bytes land in the same places — so this is a pure speed knob, unlike the ones
// below it. Default off until the on-device A/B is confirmed on a cool phone.
val oDirectZeroCopy: Boolean = false,
val overlap: Boolean = true, // read the next experts while the current layer computes
val denseWeights: DenseWeights = DenseWeights.ANON, // dense (non-expert) weight residency policy
val prefetchLayers: Int = 0, // temporal prefetch depth K (0 = off); needs the cache
@ -102,6 +107,7 @@ data class AppSettings(
}
a += listOf("--io-threads", ioThreads.toString())
if (!oDirect) a += "--no-odirect"
if (oDirect && oDirectZeroCopy) a += "--odirect-zero-copy"
if (overlap) a += "--overlap"
// Dense (non-expert) weight policy — one canonical flag (mmap | warm | anon).
a += listOf("--dense-weights", denseWeights.flag)
@ -132,7 +138,8 @@ data class AppSettings(
*/
fun sessionSignature(modelPath: String): String =
listOf(modelPath, mmap, cacheMb, cacheCeilMb, ioThreads, threads, nExpertUsed, sessionCtx, oDirect,
overlap, denseWeights, prefetchLayers, predictPrefetch, predictSpecMax, dropColdPct)
oDirectZeroCopy, overlap, denseWeights, prefetchLayers, predictPrefetch, predictSpecMax,
dropColdPct)
.joinToString("|")
fun save(ctx: Context) {
@ -142,6 +149,7 @@ data class AppSettings(
.putInt("ioThreads", ioThreads).putInt("threads", threads)
.putInt("nExpertUsed", nExpertUsed)
.putInt("nPredict", nPredict).putBoolean("oDirect", oDirect)
.putBoolean("oDirectZeroCopy", oDirectZeroCopy)
.putBoolean("overlap", overlap)
.putString("denseWeights", denseWeights.name)
.putInt("prefetchLayers", prefetchLayers)
@ -250,6 +258,7 @@ data class AppSettings(
nExpertUsed = p.getInt("nExpertUsed", d.nExpertUsed),
nPredict = p.getInt("nPredict", d.nPredict),
oDirect = p.getBoolean("oDirect", d.oDirect),
oDirectZeroCopy = p.getBoolean("oDirectZeroCopy", d.oDirectZeroCopy),
overlap = p.getBoolean("overlap", d.overlap),
denseWeights = run {
val saved = p.getString("denseWeights", null)

View file

@ -101,6 +101,12 @@ fun SettingsScreen(current: AppSettings, onChange: (AppSettings) -> Unit, onBack
"Bypass the page cache for expert reads. Falls back to buffered automatically if unsupported",
current.oDirect, enabled = stream,
) { onChange(current.copy(oDirect = it)) }
SwitchRow(
"Zero-copy reads",
"Read expert slices straight into the cache instead of via a bounce buffer. " +
"Needs Direct I/O; same bytes, same output — purely faster",
current.oDirectZeroCopy, enabled = stream && current.oDirect,
) { onChange(current.copy(oDirectZeroCopy = it)) }
SwitchRow(
"I/O–compute overlap",
"Read the next experts while the current layer computes, hiding read latency",