From bde428964886372dd341e0dc82402df326a0235c Mon Sep 17 00:00:00 2001 From: Pascal Date: Sun, 30 Aug 2026 10:43:03 +0200 Subject: [PATCH] kv-cells: stop the sequence scan once all sequences are seen for_each_token_in tested all LLAMA_MAX_SEQ sequences for every used cell, while a cell almost always belongs to one. The scan now stops once the cell's own sequences have been seen. Same visit order, same callback arguments, so behaviour is unchanged. get_prev_tokens is the only caller, so this affects the n-gram path. RTX PRO 6000, Qwen3.8-Flash-Next UD-Q4_K_XL, fa on, warm runs: 55k context generation 56.3 -> 74.3 t/s 132k context generation 33.6 -> 50.9 t/s Prompt processing is unchanged, the scan is amortised over the ubatch there. The gain follows the number of used cells, so it grows with context and is invisible on short prompts. --- src/llama-kv-cells.h | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/src/llama-kv-cells.h b/src/llama-kv-cells.h index 5167c037db3e..a4292c79e443 100644 --- a/src/llama-kv-cells.h +++ b/src/llama-kv-cells.h @@ -320,13 +320,14 @@ class llama_kv_cells { } const auto m = seq[i] & seqs; - if (m.none()) { - continue; - } - for (llama_seq_id s = 0; s < LLAMA_MAX_SEQ; ++s) { + // a cell carries a handful of sequences at most, out of LLAMA_MAX_SEQ + size_t left = m.count(); + + for (llama_seq_id s = 0; left > 0 && s < (llama_seq_id) LLAMA_MAX_SEQ; ++s) { if (m.test(s)) { f(s, pos[i], ext[i].tok); + --left; } } }