From 08fcb468b6e160c25788d0d5311b1e7034c8bbc6 Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Tue, 22 Sep 2026 21:11:01 -0300 Subject: [PATCH 1/2] docs(streaming): document warm_willneed as inert under prod_k8() While investigating #110 (edge0-8b first-token latency), traced why toggling warm_willneed had no visible effect on that tier: it's only read by StreamingSwitchGLU.prefetch() and stage_experts(), and stage_experts() early-returns immediately when staged is False. prod_k8() (edge0-8b's production profile) sets staged=False and history_prefetch=False, so neither call site that reads warm_willneed ever executes -- the flag is dead code for that tier's default config, not merely ineffective in this instance. Documents this in both the class docstring (options.py) and the streaming.md options table, so the next person investigating slow first-token latency on edge0-8b doesn't spend time on the same dead end. Co-Authored-By: Claude Sonnet 5 --- docs/streaming.md | 2 +- src/edge0/streaming/options.py | 5 +++++ 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/docs/streaming.md b/docs/streaming.md index c049da0..8719194 100644 --- a/docs/streaming.md +++ b/docs/streaming.md @@ -73,7 +73,7 @@ With `use_compile`, the staged/exact paths are wrapped in `mx.compile`: | `load_threads` / `prefetch_threads` | Build / prefetch thread counts | 8 / 4 | | `full_layer_prefill` / `prefill_full_layers` | Whole-layer prefill loading / number of leading layers | False / 0 | | `prefill_hot` | hot stack size during prefill | 0 | -| `warm_willneed` | Kernel bulk readahead (`madvise WILLNEED`) over the expert ranges a prefetch/stage is about to touch | False | +| `warm_willneed` | Kernel bulk readahead (`madvise WILLNEED`) over the expert ranges a prefetch/stage is about to touch. Only read by `prefetch()` and `stage_experts()` -- **inert whenever `staged=False` and `history_prefetch=False`**, which is `prod_k8()`'s (edge0-8b) default, so toggling it has no effect on that tier without also enabling one of those paths. | False | | `use_compile` / `top_k` | compile wrapping / routing top-k override | True / None | Presets: `staged_k4()` (edge0-35b: staged decode with 4 slots, prefill hot stack 32, on-demand prefill), `prod_k8()` (edge0-8b: the reference deployment profile — staged decode off, E3b whole-layer prefill), and `staged_k8()` (the plain K=8 staged variant). Both tiers share `cache_slots=64`. diff --git a/src/edge0/streaming/options.py b/src/edge0/streaming/options.py index 8a293f8..44d005b 100644 --- a/src/edge0/streaming/options.py +++ b/src/edge0/streaming/options.py @@ -90,6 +90,11 @@ class LayerOptions: #: a step otherwise degrades into "cold pages x per-fault latency", with #: those faults serializing on the VM map lock. Retains no MLX arrays, #: so it does not displace the page cache. + #: Only consulted by ``prefetch()`` and ``stage_experts()`` — INERT when + #: both ``staged`` and ``history_prefetch`` are False, since neither call + #: site then ever runs. ``prod_k8()`` (edge0-8b's production profile) is + #: exactly that combination, so toggling this flag on that tier has no + #: effect (see issue #110). warm_willneed: bool = False # ---- presets ---------------------------------------------------------- From 2c609e3b0e44f93ee8589df6da9f8c83d2681fb3 Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Sun, 4 Oct 2026 15:57:08 -0300 Subject: [PATCH 2/2] docs(streaming): qualify prefill readahead prerequisites --- docs/streaming.md | 9 ++++++--- python/src/edge0/streaming/options.py | 8 +++++--- 2 files changed, 11 insertions(+), 6 deletions(-) diff --git a/docs/streaming.md b/docs/streaming.md index 4466cec..fcb4879 100644 --- a/docs/streaming.md +++ b/docs/streaming.md @@ -81,9 +81,12 @@ Presets: `staged_k4()` (edge0-35b: staged decode with 4 slots, prefill hot stack `warm_willneed` is consulted only by `prefetch()` when there are missing experts and by `stage_experts()` on staged layers. Turning it on does not enable either path. Disabling history prefetch and staging removes those automatic decode -paths, but explicit prefetch calls (including `prefetch_from_prefill()`) can -still issue readahead. The flag does not warm plain on-demand or whole-layer -loads by itself, so it is not a general first-token-latency switch (see #110). +paths, but a direct `prefetch(experts)` call can still issue readahead for +missing experts. `prefetch_from_prefill()` calls `prefetch()` only if a prefill +expert set was captured while staging was enabled; with staging disabled from +initialization, it is a no-op. The flag does not warm plain on-demand or +whole-layer loads by itself, so it is not a general first-token-latency switch +(see #110). The whole-layer prefill is the fastest path **when the checkpoint stays in the page cache** (warm 27-token prefill: 0.24 s vs 0.37 s on-demand on an M4 Pro), and the slowest one when it does not (cold: 5.3 s / 4.06 GiB read vs 0.6-1.1 s / 0.4-0.8 GiB; the on-demand figure varies with how many distinct experts the prompt routes to). `edge0 demo|chat|serve --prefill-ondemand` selects the on-demand path for machines in the second group. diff --git a/python/src/edge0/streaming/options.py b/python/src/edge0/streaming/options.py index 25da43e..6b12fe5 100644 --- a/python/src/edge0/streaming/options.py +++ b/python/src/edge0/streaming/options.py @@ -92,9 +92,11 @@ class LayerOptions: #: so it does not displace the page cache. #: Only consulted by ``prefetch()`` (for missing experts) and #: ``stage_experts()`` (on staged layers). Does not enable either path - #: or affect plain on-demand / whole-layer loading. Explicit prefetch - #: calls, including ``prefetch_from_prefill()``, can still consult it - #: when automatic history prefetch and staging are disabled. + #: or affect plain on-demand / whole-layer loading. Direct + #: ``prefetch(experts)`` calls can still consult it when automatic + #: history prefetch and staging are disabled. ``prefetch_from_prefill()`` + #: requires an expert set captured while staging was enabled; with + #: staging disabled from initialization, that helper is a no-op. warm_willneed: bool = False # ---- presets ----------------------------------------------------------