From 58014c472c28cd00243dc86d8027fff45034c793 Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Tue, 22 Sep 2026 17:21:24 -0300 Subject: [PATCH 1/2] docs(streaming): correct load_hot_layer's "page cache" backing claim load_hot_layer()'s docstring described the hot-expert backing store as "page cache" that "survives across requests, counts ZERO toward MLX active" -- true of the zero-MLX-cost and survives-across-requests parts, but the backing is built with np.concatenate(rows), which copies. That makes it resident anonymous process memory, not mmap'd/file-backed page cache: it is not reclaimable by the OS the way page cache is, and does not shrink under memory pressure the way the docstring's "page cache" framing implies. Fixes the misleading description and the matching inline comment further down the same method. No behavior change. Addresses point 2 of #106 (the README wording in points 1 and 3 was already addressed in the issue thread by the maintainer). Co-Authored-By: Claude Sonnet 5 --- python/src/edge0/streaming/layer.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/python/src/edge0/streaming/layer.py b/python/src/edge0/streaming/layer.py index ef1320b..b2b2d99 100644 --- a/python/src/edge0/streaming/layer.py +++ b/python/src/edge0/streaming/layer.py @@ -992,8 +992,11 @@ def load_hot_layer(self, n_hot: int = 128) -> None: """Load/refresh the layer's hot-expert weights. Backing store = one NUMPY array per (proj, part) holding the - stacked rows of the top-n_hot experts (page cache, survives across - requests, counts ZERO toward MLX active). The mx window + stacked rows of the top-n_hot experts. ``np.concatenate`` copies, + so this is resident anonymous process memory, not page cache -- + it survives across requests (the layer keeps the reference) and + counts ZERO toward MLX active, but it is not reclaimable by the + OS the way mmap'd file-backed pages are. The mx window (materialize_hot) exposes a few layers at a time so peak MLX memory stays at window_size * n_hot * ~1.2MB instead of 40 * n_hot * 1.2MB. @@ -1034,7 +1037,8 @@ def load_hot_layer(self, n_hot: int = 128) -> None: raw = self._shard_for(name).raw(name) per = raw.size // self.num_experts rows = [raw[e * per:(e + 1) * per] for e in hot] - # concatenated numpy backing (page cache, not GPU), plus + # concatenated numpy backing (anonymous process memory, + # not GPU -- see load_hot_layer's docstring), plus # one all-zero row: the prefill path sends misses to row # n_hot ("overflow row") and needs it to contribute zero. # Without it that gather reads past the end of the stack. From 538b6695b709188d9c8bfd7062782637ec68791b Mon Sep 17 00:00:00 2001 From: Demetrios Agourakis Date: Sun, 4 Oct 2026 11:36:56 -0300 Subject: [PATCH 2/2] docs(streaming): distinguish paging from file-backed eviction --- python/src/edge0/streaming/layer.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/python/src/edge0/streaming/layer.py b/python/src/edge0/streaming/layer.py index b2b2d99..a17c247 100644 --- a/python/src/edge0/streaming/layer.py +++ b/python/src/edge0/streaming/layer.py @@ -993,10 +993,11 @@ def load_hot_layer(self, n_hot: int = 128) -> None: Backing store = one NUMPY array per (proj, part) holding the stacked rows of the top-n_hot experts. ``np.concatenate`` copies, - so this is resident anonymous process memory, not page cache -- - it survives across requests (the layer keeps the reference) and - counts ZERO toward MLX active, but it is not reclaimable by the - OS the way mmap'd file-backed pages are. The mx window + so this is anonymous process memory, not file-backed page cache. + The layer retains it across requests; these host allocations are + outside MLX active memory. The OS may page them out, but cannot + discard and reload them from the checkpoint like clean mapped + file pages. The mx window (materialize_hot) exposes a few layers at a time so peak MLX memory stays at window_size * n_hot * ~1.2MB instead of 40 * n_hot * 1.2MB. @@ -1037,8 +1038,7 @@ def load_hot_layer(self, n_hot: int = 128) -> None: raw = self._shard_for(name).raw(name) per = raw.size // self.num_experts rows = [raw[e * per:(e + 1) * per] for e in hot] - # concatenated numpy backing (anonymous process memory, - # not GPU -- see load_hot_layer's docstring), plus + # copied numpy backing (anonymous process memory), plus # one all-zero row: the prefill path sends misses to row # n_hot ("overflow row") and needs it to contribute zero. # Without it that gather reads past the end of the stack.