From 3b8acea0bf02f159bdd4c23750bb3c39d9d0f33a Mon Sep 17 00:00:00 2001 From: Kyle Tse Date: Fri, 2 Oct 2026 11:36:03 +0000 Subject: [PATCH 1/4] perf: keyword-first hybrid fusion, query-side embedding without loading the model, hybrid >= keyword CI floor --- CHANGELOG.md | 3 + bench/run.py | 12 +- bench/test_run.py | 15 +- crates/lockdocs-core/src/index.rs | 2 +- crates/lockdocs-core/src/query.rs | 126 ++++++++----- crates/lockdocs-core/src/semantic.rs | 255 ++++++++++++++++++++++++++- docs/benchmarks.md | 2 +- docs/guide/how-it-works.md | 2 +- 8 files changed, 364 insertions(+), 53 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 945977c..ebe5a25 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,9 @@ ## 0.5.0 - **First-use fetch no longer depends on GitHub's REST API quota.** Upstream docs are found with `git ls-remote`-style tag lookups and read from one streamed `codeload.github.com` tarball per repository, filtered to docs paths, instead of the REST tree and per-file requests (60 per hour per IP anonymously; 2 of 10 benchmark projects failed in our clean-quota run, up to 6 of 10 behind a shared IP). Cold-cache, no token: 10/10 on the 10-question version set, no 403, 1-5 GitHub connections per project instead of 2-28. `upstream::FORMAT` is 5. A missing tag error lists the tag names tried. An archive over 150 MB compressed for explicit fetch, or 64 MB for the automatic first-use fetch (or a codeload failure, or too little of the automatic 45 s budget left), falls back to the per-file REST path for that repository only, and the manifest records `via: codeload` or `rest`. Measured (cold cache, no token, 1 s pacing between GitHub connections, same 10-question set): before 8/10 correct, 2 projects hit 403, 152 GitHub connections (34 REST, 118 raw), 145 s total fetch time, 1.8 MB down; after 10/10, 0 403s, 31 connections (17 codeload, none REST or raw), 27 s, 56 MB down. Per package, next 14.2.35 and 15.1.0 download a 42-46 MB tarball in 2.7-5 s and TypeScript 5.6.3 a 32 MB one in 8 s, each over 2 connections, all through codeload. Explicit docs-site fetches also download whole-repository tarballs. `GITHUB_TOKEN`/`GH_TOKEN` stays on api.github.com for explicit fetch only, as in 0.4.0; git, codeload and raw requests carry no credential, so a private repository fails with "private repository: not supported". Caches of format 4 stay usable by automatic and offline queries (an explicit fetch still refreshes them). Trade-off: bytes go up (the tarball holds the whole repository) while connections and quota use go down. +- **Hybrid search is no longer worse than keyword search.** The fusion is now keyword-first: weighted reciprocal rank (90% keyword ranking with all docs boosts, 10% embedding ranking) replaces the min-max score mix that let the embedding outvote exact API-name matches, and the embedding is skipped when one entry leads by at least 2x. On the 105-question benchmark (package files, no upstream) hybrid goes from 60 to 64 of 105 against 59 for keyword-only, with the held-out set at 8/18 (was 6/18, keyword 8/18). +- **Hybrid search is as light as keyword search.** A query reads only its own tokens' rows from the model file (positioned reads) instead of loading the 32 MB weight table: median CLI peak RSS 48 MB to 18 MB, median latency 42 ms to 18 ms (keyword-only 12 ms). The full model loads only when an index is built. +- **CI floor.** The benchmark job fails when hybrid scores below keyword-only on the same docs (package-only or prefetched); the package-only hybrid floor rises from 60 to 62. ## 0.4.0 diff --git a/bench/run.py b/bench/run.py index 22bc82a..ddc7374 100755 --- a/bench/run.py +++ b/bench/run.py @@ -140,6 +140,8 @@ def run_context7(q, version): ("first-use-default", "lockdocs, real first-use defaults (empty isolated cache, anonymous release-tag fetch)", {"LOCKDOCS_FETCH": None, "LOCKDOCS_NO_UPSTREAM": None, "LOCKDOCS_EMBED": None, "LOCKDOCS_OFFLINE": None, "GITHUB_TOKEN": None, "GH_TOKEN": None}), ("fetched", "lockdocs, hybrid + upstream docs (after `lockdocs fetch`)", {"LOCKDOCS_FETCH": "1"}), + # After "fetched": it reuses the docs that variant's prefetch cached. + ("fetched-keyword", "lockdocs, keyword only (BM25) + upstream docs", {"LOCKDOCS_EMBED": "0", "LOCKDOCS_FETCH": "1"}), ] @@ -250,9 +252,17 @@ def main(): def check_floors(summary, total): if total != 105: return - for variant, floor in [("fetched", 96), ("hybrid", 60)]: + for variant, floor in [("fetched", 96), ("hybrid", 62)]: if variant in summary and summary[variant]["passed"] < floor: raise RuntimeError(f"{variant} regressed: {summary[variant]['passed']}/105, required >= {floor}/105") + # Hybrid retrieval exists to beat keyword search; if it scores lower on the + # same docs, the fusion is hurting and the job fails. + for hybrid, keyword in [("hybrid", "keyword"), ("fetched", "fetched-keyword")]: + if hybrid in summary and keyword in summary and summary[hybrid]["passed"] < summary[keyword]["passed"]: + raise RuntimeError( + f"{hybrid} ({summary[hybrid]['passed']}/105) scores below {keyword} ({summary[keyword]['passed']}/105): " + "embedding fusion must never lose to keyword-only search" + ) def agg(rs, total): diff --git a/bench/test_run.py b/bench/test_run.py index 2acd819..bde2b41 100644 --- a/bench/test_run.py +++ b/bench/test_run.py @@ -18,12 +18,23 @@ class BenchmarkTests(unittest.TestCase): def test_floors_apply_only_to_full_suite(self): - runner.check_floors({"fetched": {"passed": 96}, "hybrid": {"passed": 60}}, 105) + runner.check_floors({"fetched": {"passed": 96}, "hybrid": {"passed": 62}}, 105) runner.check_floors({"fetched": {"passed": 0}}, 1) - for variant, score in [("fetched", 95), ("hybrid", 59)]: + for variant, score in [("fetched", 95), ("hybrid", 61)]: with self.assertRaises(RuntimeError): runner.check_floors({variant: {"passed": score}}, 105) + def test_hybrid_never_scores_below_keyword(self): + ok = {"keyword": {"passed": 62}, "hybrid": {"passed": 62}, "fetched-keyword": {"passed": 96}, "fetched": {"passed": 97}} + runner.check_floors(ok, 105) + runner.check_floors({"keyword": {"passed": 70}, "hybrid": {"passed": 69}}, 1) # partial runs are not gated + for bad in [ + {"keyword": {"passed": 63}, "hybrid": {"passed": 62}}, + {"fetched-keyword": {"passed": 98}, "fetched": {"passed": 97}}, + ]: + with self.assertRaises(RuntimeError): + runner.check_floors(bad, 105) + def test_fetch_formatter_preserves_flat_and_nested_file_counts(self): for package in [ {"package": "axum@0.7.9", "files": 20}, diff --git a/crates/lockdocs-core/src/index.rs b/crates/lockdocs-core/src/index.rs index ff8fa10..06230e5 100644 --- a/crates/lockdocs-core/src/index.rs +++ b/crates/lockdocs-core/src/index.rs @@ -238,7 +238,7 @@ pub fn build(dep: &Dep, src: &Source, root: &Path, up: Option<&(PathBuf, Manifes /// Load from the disk cache, or build and store. pub fn load_or_build(dep: &Dep, src: &Source, root: &Path, up: Option<&(PathBuf, Manifest)>) -> PackageIndex { let types = types_pkg(dep, root); - let embed_id = if embed::get().is_some() { embed::MODEL_ID } else { "" }; + let embed_id = if embed::available() { embed::MODEL_ID } else { "" }; let path = cache_path(dep, src, types.as_deref(), up, embed_id); if let Ok(bytes) = std::fs::read(&path) { if let Ok(idx) = postcard::from_bytes::(&bytes) { diff --git a/crates/lockdocs-core/src/query.rs b/crates/lockdocs-core/src/query.rs index 95bab72..5a58008 100644 --- a/crates/lockdocs-core/src/query.rs +++ b/crates/lockdocs-core/src/query.rs @@ -17,8 +17,17 @@ use std::sync::{Arc, Mutex}; pub const DEFAULT_TOKENS: usize = 1200; /// Results scoring below this fraction of the best are left out (after 3). const RELEVANCE_FLOOR: f32 = 0.3; -/// Share of the fused score from the embedding similarity. -const DENSE_WEIGHT: f32 = 0.45; +/// Share of the fused rank score from the embedding ranking; the rest is the +/// keyword ranking. Small on purpose: measured on the 105-question benchmark, +/// anything above ~0.15 lets the embedding outvote exact API-name matches. +const DENSE_WEIGHT: f32 = 0.1; +/// Reciprocal-rank constant: `1 / (RRF_K + rank)`. +const RRF_K: f32 = 10.0; +/// Embedding candidates considered per query. +const DENSE_POOL: usize = 300; +/// The keyword leader is final when it scores at least this many times the +/// runner-up (after boosts); the embedding is not even looked at. +const DECISIVE_RATIO: f32 = 2.0; /// Share of the keyword score from headings and first sentences. const HEAD_WEIGHT: f32 = 0.25; /// Cross-dependency searches index at most this many direct dependencies. @@ -257,7 +266,7 @@ impl Engine { let key = format!("{}:{}@{}:{}", dep.eco, dep.name, src.version, src.dir.display()); if let Some(i) = self.indexes.lock().unwrap().get(&key) { // Rebuild once the embedding model has arrived. - if !i.embed.is_empty() || embed::get().is_none() { + if !i.embed.is_empty() || !embed::available() { if let Some(n) = self.upstream_notes.lock().unwrap().get(&dep.id()) { note = Some(match note { Some(old) => format!("{old} {n}"), @@ -1056,8 +1065,9 @@ fn api_words(ready: &[ReadyPkg], q: &str) -> Vec { out } -/// BM25 and embedding similarity fused (each normalized to its best hit), -/// then docs-specific boosts, then deprecation redirects ("use X instead") +/// Keyword ranking (BM25 over text and headings, then docs-specific boosts) +/// first; when no entry clearly leads, the embedding ranking is fused in by +/// weighted reciprocal rank. Then deprecation redirects ("use X instead") /// lift the API they point to. Returns (score, package, entry), best first. fn hybrid_rank( ready: &[ReadyPkg], @@ -1068,8 +1078,8 @@ fn hybrid_rank( idents: &[String], changes: bool, ) -> Vec<(f32, usize, usize)> { - // (full-text BM25, heading BM25, dense), each normalized to its best hit. - let mut fused: HashMap = HashMap::new(); + // (full-text BM25, heading BM25), each normalized to its best hit. + let mut fused: HashMap = HashMap::new(); let expanded = bm25::expand(qterms); let bm_hits = bm.search_weighted(&expanded); let bmax = bm_hits.first().map_or(1.0, |h| h.score).max(1e-6); @@ -1082,59 +1092,83 @@ fn hybrid_rank( for h in h_hits.iter().take(1500) { fused.entry(h.doc as usize).or_default().1 = h.score / hmax; } - let dense_w = std::env::var("LOCKDOCS_DENSE_WEIGHT").ok().and_then(|v| v.parse().ok()).unwrap_or(DENSE_WEIGHT); - let qv = if ready.iter().all(|r| !r.0.vecs.is_empty()) { - embed::get().and_then(|m| m.embed(q)) + let dense_w: f32 = std::env::var("LOCKDOCS_DENSE_WEIGHT").ok().and_then(|v| v.parse().ok()).unwrap_or(DENSE_WEIGHT); + // Query words that are rare in these packages; common ones ("futures" in + // tokio) say little about which entry is meant. + let n = refs.len().max(1) as f32; + let rare: Vec = qterms.iter().filter(|t| (bm.df(t) as f32) < n * 0.03).cloned().collect(); + let weight = |i: usize| { + let (pi, ei) = refs[i]; + let e = &ready[pi].0.entries[ei]; + boost(e, idents, changes, major_of(&ready[pi].0.version)) * name_hit(e, &rare) + }; + // Keyword ranking first, with the docs-specific boosts: this is what the + // answer is built on. + let mut lex: Vec<(usize, f32)> = fused.iter().map(|(&i, &(b, h))| (i, ((1.0 - head_w) * b + head_w * h) * weight(i))).collect(); + lex.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal).then(a.0.cmp(&b.0))); + // A clear leader needs no second opinion: skip the embedding entirely. + let decisive = match (lex.first(), lex.get(1)) { + (Some(t), Some(s)) => t.1 >= s.1 * DECISIVE_RATIO, + (Some(_), None) => true, + _ => false, + }; + let debug = std::env::var("LOCKDOCS_DEBUG").is_ok(); + let qv = if !decisive && dense_w > 0.0 && ready.iter().all(|r| !r.0.vecs.is_empty()) { + embed::query_model().and_then(|m| m.embed(q)) } else { None }; + // Embedding rank of each entry among the 300 nearest to the query. + let mut dense_rank: HashMap = HashMap::new(); if let Some(qv) = &qv { let mut sims: Vec<(usize, f32)> = refs .par_iter() .enumerate() .map(|(i, &(pi, ei))| (i, embed::cosine(qv, &ready[pi].0.vecs[ei]))) .collect(); - sims.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal)); - let cmax = sims.first().map_or(1.0, |s| s.1); - let floor = sims.get(300).map_or(0.0, |s| s.1); - if std::env::var("LOCKDOCS_DEBUG").is_ok() { - eprintln!("dense cmax={cmax:.3} floor={floor:.3}"); - for (rank, (i, c)) in sims.iter().enumerate() { + sims.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal).then(a.0.cmp(&b.0))); + for (r, (i, _)) in sims.iter().take(DENSE_POOL).enumerate() { + dense_rank.insert(*i, r); + } + if debug { + for (i, c) in sims.iter().take(10) { let (pi, ei) = refs[*i]; - if rank < 10 { - eprintln!("dense rank {rank} cos={c:.3} {}", ready[pi].0.entries[ei].path); - } + eprintln!("dense cos={c:.3} {}", ready[pi].0.entries[ei].path); } } - for (i, c) in sims.iter().take(300) { - fused.entry(*i).or_default().2 = ((c - floor) / (cmax - floor).max(1e-6)).max(0.0); + } + if debug { + eprintln!("decisive={decisive} dense={}", qv.is_some()); + } + // Reciprocal-rank fusion, weighted towards the keyword ranking (the + // embedding breaks near-ties and rescues paraphrases, it cannot outvote + // an exact name or heading match). Scores are scaled so the best is 1. + let rrf = |r: usize| 1.0 / (RRF_K + r as f32); + let mut scored: Vec<(f32, usize, usize)> = if qv.is_some() { + let mut ids: HashSet = lex.iter().map(|l| l.0).collect(); + ids.extend(dense_rank.keys().copied()); + let lex_rank: HashMap = lex.iter().enumerate().map(|(r, l)| (l.0, r)).collect(); + ids.into_iter() + .map(|i| { + let l = lex_rank.get(&i).map_or(0.0, |&r| rrf(r)); + let d = dense_rank.get(&i).map_or(0.0, |&r| rrf(r)); + let (pi, ei) = refs[i]; + ((1.0 - dense_w) * l + dense_w * d, pi, ei) + }) + .collect() + } else { + lex.iter().map(|&(i, s)| (s, refs[i].0, refs[i].1)).collect() + }; + if qv.is_some() { + let top = scored.iter().map(|s| s.0).fold(0.0f32, f32::max).max(1e-9); + scored.iter_mut().for_each(|s| s.0 /= top); + } + if debug { + scored.sort_by(|a, b| b.0.partial_cmp(&a.0).unwrap_or(std::cmp::Ordering::Equal)); + for (s, pi, ei) in scored.iter().take(10) { + eprintln!("{s:.3} {}", ready[*pi].0.entries[*ei].path); } } - let w = if qv.is_some() { dense_w } else { 0.0 }; - // Query words that are rare in these packages; common ones ("futures" in - // tokio) say little about which entry is meant. - let n = refs.len().max(1) as f32; - let rare: Vec = qterms.iter().filter(|t| (bm.df(t) as f32) < n * 0.03).cloned().collect(); - let debug = std::env::var("LOCKDOCS_DEBUG").is_ok(); - let mut scored: Vec<(f32, usize, usize)> = fused - .into_iter() - .map(|(i, (b, h, d))| { - let (pi, ei) = refs[i]; - let e = &ready[pi].0.entries[ei]; - let lexical = (1.0 - head_w) * b + head_w * h; - let major = major_of(&ready[pi].0.version); - let s = ((1.0 - w) * lexical + w * d) * boost(e, idents, changes, major) * name_hit(e, &rare); - if debug && s > 0.3 { - eprintln!( - "{s:.3} bm={b:.3} head={h:.3} dense={d:.3} boost={:.2} name={:.2} {}", - boost(e, idents, changes, major), - name_hit(e, &rare), - e.path - ); - } - (s, pi, ei) - }) - .collect(); scored.sort_by(|a, b| b.0.partial_cmp(&a.0).unwrap_or(std::cmp::Ordering::Equal)); // Deprecation redirects among the top results. let mut lifted: Vec<(f32, usize, String)> = Vec::new(); diff --git a/crates/lockdocs-core/src/semantic.rs b/crates/lockdocs-core/src/semantic.rs index 38c67be..54f3b7e 100644 --- a/crates/lockdocs-core/src/semantic.rs +++ b/crates/lockdocs-core/src/semantic.rs @@ -3,7 +3,8 @@ use anyhow::Result; use mcp_kit::embed::{self, Model, Tokenization, POTION_RETRIEVAL_32M}; -use std::path::PathBuf; +use std::collections::HashMap; +use std::path::{Path, PathBuf}; use std::sync::{Arc, OnceLock, RwLock}; pub use mcp_kit::embed::{cosine, quantize, Vec8}; @@ -22,6 +23,228 @@ pub fn installed() -> bool { embed::installed_at(&model_dir()) } +/// Query-side embedder: reads only the vocabulary, the row scales and the rows +/// of the query's own tokens (about 512 bytes each) with positioned reads, so +/// a search never pulls the 32 MB weight table into memory. It reproduces +/// `Model::embed` under `Tokenization::Identifiers` exactly (a test checks +/// this against the full model); entry vectors are still built by the full +/// model, only once per package version. +pub struct QueryModel { + file: std::fs::File, + vocab: PathBuf, + rows: usize, + dims: usize, +} + +const MAX_WORD_CHARS: usize = 100; + +/// FxHash for short vocabulary keys (SipHash costs milliseconds over 63k lines). +#[derive(Default)] +struct Fx(u64); + +impl std::hash::Hasher for Fx { + fn write(&mut self, bytes: &[u8]) { + for chunk in bytes.chunks(8) { + let mut b = [0u8; 8]; + b[..chunk.len()].copy_from_slice(chunk); + self.0 = (self.0.rotate_left(5) ^ u64::from_le_bytes(b)).wrapping_mul(0x51_7c_c1_b7_27_22_0a_95); + } + } + fn write_u8(&mut self, i: u8) { + self.0 = (self.0.rotate_left(5) ^ i as u64).wrapping_mul(0x51_7c_c1_b7_27_22_0a_95); + } + fn finish(&self) -> u64 { + self.0 + } +} + +type Pieces = HashMap>; + +impl QueryModel { + pub fn open(dir: &Path) -> Option { + let file = std::fs::File::open(dir.join("model.q8")).ok()?; + let mut head = [0u8; 8]; + read_at(&file, &mut head, 0).ok()?; + let rows = u32::from_le_bytes(head[0..4].try_into().ok()?) as usize; + let dims = u32::from_le_bytes(head[4..8].try_into().ok()?) as usize; + (file.metadata().ok()?.len() as usize == 8 + rows * 4 + rows * dims).then(|| QueryModel { + file, + vocab: dir.join("vocab.txt"), + rows, + dims, + }) + } + + /// The vocabulary entries that can spell any of `words`: every substring + /// of a word is a candidate, so one pass over vocab.txt finds them all + /// without building the 63k-entry lookup tables. + fn pieces(&self, words: &[String]) -> Option<(Pieces, Pieces)> { + let mut first = Pieces::default(); + let mut cont = Pieces::default(); + for w in words { + if w.chars().count() > MAX_WORD_CHARS { + continue; + } + let bounds: Vec = w.char_indices().map(|(i, _)| i).chain([w.len()]).collect(); + for (si, &s) in bounds.iter().enumerate() { + for &e in &bounds[si + 1..] { + let map = if s == 0 { &mut first } else { &mut cont }; + map.entry(w[s..e].to_string()).or_insert(u32::MAX); + } + } + } + let text = std::fs::read_to_string(&self.vocab).ok()?; + let mut n = 0; + for (i, w) in text.split('\n').enumerate() { + n += 1; + match w.strip_prefix("##") { + Some(rest) if !rest.is_empty() => { + if let Some(slot) = cont.get_mut(rest) { + *slot = i as u32; + } + } + _ => { + if let Some(slot) = first.get_mut(w) { + *slot = i as u32; + } + } + } + } + (n == self.rows).then_some((first, cont)) + } + + fn wordpiece(first: &Pieces, cont: &Pieces, word: &str, ids: &mut Vec) { + if word.chars().count() > MAX_WORD_CHARS { + return; + } + let start_len = ids.len(); + let mut start = 0; + while start < word.len() { + let map = if start == 0 { first } else { cont }; + let mut end = word.len(); + let mut found = None; + while end > start { + if word.is_char_boundary(end) { + if let Some(id) = map.get(&word[start..end]).filter(|id| **id != u32::MAX) { + found = Some(*id); + break; + } + } + end -= 1; + } + match found { + Some(id) => { + ids.push(id); + start = end; + } + None => { + ids.truncate(start_len); + return; + } + } + } + } + + /// Unit-length embedding of a query, as `Model::embed` gives it. + pub fn embed(&self, text: &str) -> Option> { + let words = identifier_words(text); + let (first, cont) = self.pieces(&words)?; + let mut ids = Vec::new(); + for word in &words { + Self::wordpiece(&first, &cont, word, &mut ids); + } + if ids.is_empty() { + return None; + } + let mut acc = vec![0f32; self.dims]; + let mut row = vec![0u8; self.dims]; + let mut scale = [0u8; 4]; + let base = 8 + self.rows as u64 * 4; + for id in ids { + let r = id as usize; + read_at(&self.file, &mut scale, 8 + r as u64 * 4).ok()?; + read_at(&self.file, &mut row, base + (r * self.dims) as u64).ok()?; + let s = f32::from_le_bytes(scale); + for (a, q) in acc.iter_mut().zip(&row) { + *a += *q as i8 as f32 * s; + } + } + let norm = acc.iter().map(|x| x * x).sum::().sqrt(); + if norm < 1e-9 { + return None; + } + acc.iter_mut().for_each(|x| *x /= norm); + Some(acc) + } +} + +#[cfg(unix)] +fn read_at(f: &std::fs::File, buf: &mut [u8], off: u64) -> std::io::Result<()> { + std::os::unix::fs::FileExt::read_exact_at(f, buf, off) +} + +#[cfg(windows)] +fn read_at(f: &std::fs::File, buf: &mut [u8], mut off: u64) -> std::io::Result<()> { + let mut done = 0; + while done < buf.len() { + let n = std::os::windows::fs::FileExt::seek_read(f, &mut buf[done..], off)?; + if n == 0 { + return Err(std::io::ErrorKind::UnexpectedEof.into()); + } + done += n; + off += n as u64; + } + Ok(()) +} + +/// Words of an identifier-aware text: camelCase and underscores split, +/// lowercased, every punctuation mark its own word (as `Tokenization::Identifiers`). +fn identifier_words(text: &str) -> Vec { + let mut spaced = String::with_capacity(text.len() + 16); + let mut prev: Option = None; + for c in text.chars() { + if let Some(p) = prev { + if c.is_uppercase() && p.is_lowercase() { + spaced.push(' '); + } + } + spaced.push(if c == '_' { ' ' } else { c }); + prev = Some(c); + } + let mut out = Vec::new(); + let mut cur = String::new(); + for c in spaced.chars() { + if c.is_whitespace() || c.is_control() { + if !cur.is_empty() { + out.push(std::mem::take(&mut cur)); + } + } else if c.is_ascii_punctuation() || (!c.is_alphanumeric() && !c.is_whitespace()) { + if !cur.is_empty() { + out.push(std::mem::take(&mut cur)); + } + out.push(c.to_string()); + } else { + cur.extend(c.to_lowercase()); + } + } + if !cur.is_empty() { + out.push(cur); + } + out +} + +/// A cheap embedder for query text, without loading the weight table. +pub fn query_model() -> Option<&'static QueryModel> { + static QM: OnceLock> = OnceLock::new(); + QM.get_or_init(|| if enabled() && installed() { QueryModel::open(&model_dir()) } else { None }) + .as_ref() +} + +/// Embeddings are on and the model is installed (nothing is loaded). +pub fn available() -> bool { + enabled() && installed() +} + static MODEL: OnceLock>>> = OnceLock::new(); /// Load existing model.q8/vocab.txt directly. The shared input policy keeps @@ -58,6 +281,36 @@ pub fn ensure() -> Result<()> { mod tests { use super::*; + #[test] + fn query_model_matches_the_full_model() { + // A tiny model: 12 word pieces, 6 dims, deterministic weights. + let words = ["use", "state", "form", "action", "##s", "get", "(", ")", "a", "##b", "context", "provider"]; + let dims = 6usize; + let dir = tempfile::tempdir().unwrap(); + let mut bytes = Vec::new(); + bytes.extend((words.len() as u32).to_le_bytes()); + bytes.extend((dims as u32).to_le_bytes()); + for i in 0..words.len() { + bytes.extend((0.01 + i as f32 * 0.003).to_le_bytes()); + } + for i in 0..words.len() { + for d in 0..dims { + bytes.push((((i * 7 + d * 13) % 255) as i32 - 127) as i8 as u8); + } + } + std::fs::write(dir.path().join("model.q8"), bytes).unwrap(); + std::fs::write(dir.path().join("vocab.txt"), words.join("\n")).unwrap(); + let full = Model::load_dir(dir.path()).unwrap().with_tokenization(Tokenization::Identifiers); + let lite = QueryModel::open(dir.path()).unwrap(); + for text in ["useState", "form_actions get(a) abs", "ContextProvider", "unknownword", "", "FormAction (use)"] { + match (full.embed(text), lite.embed(text)) { + (None, None) => {} + (Some(a), Some(b)) => assert!(a.iter().zip(&b).all(|(x, y)| (x - y).abs() < 1e-6), "{text}"), + _ => panic!("{text}: one side empty"), + } + } + } + #[test] fn persisted_entry_vectors_keep_their_postcard_layout() { #[derive(serde::Serialize)] diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 71c8f06..7d7daf0 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -6,7 +6,7 @@ - **Held-out questions:** 18 questions marked `held-out` are not used for tuning: they were written before the ranking changes they measure and have their own column. Held-out questions that are later used to diagnose a miss join the main set (their `history` field says so), and new ones replace them. - **Projects:** one project per version in [`bench/projects`](https://github.com/SylphxAI/lockdocs/tree/main/bench/projects), installed at those pins by [`bench/setup.sh`](https://github.com/SylphxAI/lockdocs/blob/main/bench/setup.sh). - **Grading:** an answer passes when it contains at least one string from every `expect` group (the version-correct API, in code or prose form) and none of the `reject` strings (the other version's API). Case-insensitive; the same grader for every tool. -- **lockdocs**, default settings (1,200-token budget), in four configurations: keyword only (`LOCKDOCS_EMBED=0`) on package files; hybrid on package files (`LOCKDOCS_FETCH=0`, upstream disabled); real first-use defaults in an initially empty isolated `LOCKDOCS_CACHE` with fetch-policy and credential variables removed, before explicit prefetch; hybrid after explicit `lockdocs fetch` added upstream docs and major-version sites. The first-use row includes cold-download latency and anonymous rate-limit failures; it is never copied from the prefetched score. Full runs enforce at least 96/105 prefetched and 60/105 package-only hybrid. Index and fetch times are reported separately. +- **lockdocs**, default settings (1,200-token budget), in four configurations: keyword only (`LOCKDOCS_EMBED=0`) on package files; hybrid on package files (`LOCKDOCS_FETCH=0`, upstream disabled); real first-use defaults in an initially empty isolated `LOCKDOCS_CACHE` with fetch-policy and credential variables removed, before explicit prefetch; hybrid after explicit `lockdocs fetch` added upstream docs and major-version sites. The first-use row includes cold-download latency and anonymous rate-limit failures; it is never copied from the prefetched score. Full runs enforce at least 96/105 prefetched and 62/105 package-only hybrid, and fail if hybrid scores below keyword-only on the same docs (package-only or prefetched). Index and fetch times are reported separately. - **Context7:** the anonymous API as its MCP server uses it: search the library, pick the top result and its listed version with the same major (exact when listed), then fetch context for the question. When that library answers HTTP 404, the next search result is tried, as an agent would. Rate-limit responses are recorded, not retried. - **Tokens:** tiktoken `o200k_base`. **Latency:** wall time per call from the same GitHub-hosted runner (for lockdocs: a fresh CLI process per question, including loading the model). - **Runner:** [`bench/run.py`](https://github.com/SylphxAI/lockdocs/blob/main/bench/run.py) via the [`bench` workflow](https://github.com/SylphxAI/lockdocs/actions/workflows/bench.yml). Reproduce: `bash bench/setup.sh && python3 bench/run.py target/release/lockdocs bench/projects out.json --fetch --context7`. diff --git a/docs/guide/how-it-works.md b/docs/guide/how-it-works.md index f3f8ee8..cd55e67 100644 --- a/docs/guide/how-it-works.md +++ b/docs/guide/how-it-works.md @@ -4,7 +4,7 @@ 2. **Locate.** Each package's files are found on disk (see [Ecosystems](./ecosystems)). If the installed version differs from the lockfile, the answer says so and shows the installed one; with fetching enabled it reads the pinned version instead. 3. **Extract.** Markdown and reStructuredText are split into heading-scoped sections (long ones at paragraph boundaries, never inside code fences). HTML headings in MDX (`

@import

`) count as headings, and `export const title` names the page. Docs pages written as React components (Tailwind's installation guides) become sections too: step titles are headings, JSX text is prose, and `code:` strings are code blocks. Source files are parsed with tree-sitter (TypeScript, Python, Rust, Go) into symbols with a signature (the declaration without its body), a cleaned doc comment, a qualified path and a line number. 4. **Index.** Each entry becomes weighted terms: names and headings count most, doc prose next, signatures and code examples least. The tokenizer keeps identifiers whole and also splits camelCase, snake_case and digits; a light stemmer (plurals, -ing, -ed, -ation/-ate, -ability/-able) and a short programming synonym table (dict/dump, parse/validate, parameter/param, JavaScript/JS, ...) bridge wording. The index is cached on disk per package, version and source location, so a version is indexed once. -5. **Rank.** BM25 over the full text, BM25 over just the heading (or name) and first sentence, so a long body does not bury what an entry says it is, and embedding similarity (a static model2vec model: a text's vector is the mean of its WordPiece token vectors, so embedding every symbol of Next.js takes well under a second), each normalized to its best hit. Then signals that matter for docs: entries whose name or heading contains a query word, identifiers written in the question (`model_dump`, `z.object`) and plain question words that name a documented top-level API of the package ("run code *after* the response" in Next.js, which exports `after`), changelog sections only when the question is about changes, and lower weight for private modules, bundled tooling, legacy copies, upgrade guides to an older major than yours ("Migrating to v6" in ESLint 9) and pages titled "(Deprecated)". Generic headings such as Parameters, Returns or Examples take their topic from the heading above them. Deprecation notes among the top results ("use `model_validate` instead") lift the API they point to. +5. **Rank.** BM25 over the full text, BM25 over just the heading (or name) and first sentence, so a long body does not bury what an entry says it is, each normalized to its best hit. Then signals that matter for docs: entries whose name or heading contains a query word, identifiers written in the question (`model_dump`, `z.object`) and plain question words that name a documented top-level API of the package ("run code *after* the response" in Next.js, which exports `after`), changelog sections only when the question is about changes, and lower weight for private modules, bundled tooling, legacy copies, upgrade guides to an older major than yours ("Migrating to v6" in ESLint 9) and pages titled "(Deprecated)". The result is the keyword ranking. When one entry clearly leads (at least twice the runner-up) it is the answer and the embedding is not consulted. Otherwise an embedding ranking (a static model2vec model: a text's vector is the mean of its WordPiece token vectors) is fused in by weighted reciprocal rank, 90% keyword and 10% embedding: the embedding breaks near-ties and rescues paraphrases, but cannot outvote an exact API-name or heading match. A query is embedded by reading only its own tokens' rows from the model file, so the 32 MB weight table never enters memory. Generic headings such as Parameters, Returns or Examples take their topic from the heading above them. Deprecation notes among the top results ("use `model_validate` instead") lift the API they point to. 6. **Pack.** The top result also quotes the first paragraph of its page and parent section (with the short list or code block that follows), when they add something: a deprecation notice at the top of the page, or the setup a subsection builds on. Results are added in rank order until the token budget is spent (the last one trimmed at a line boundary), each with a `package@version path:line` citation. `api` is exact rather than ranked: it finds entries whose name matches the last segment of the symbol, scores qualifiers (`Router` in `axum::Router::route`), follows re-exports, and adds overloads, members and other matches. From e237b5de9947943c12105d88fddf188f543a7edf Mon Sep 17 00:00:00 2001 From: Kyle Tse Date: Fri, 2 Oct 2026 13:09:08 +0000 Subject: [PATCH 2/4] fix(rank): keep the score mix when upstream docs are indexed; rank fusion only on package files (fetched 96/105) --- CHANGELOG.md | 2 +- crates/lockdocs-core/src/query.rs | 33 +++++++++++++++++++++++++++---- docs/guide/how-it-works.md | 2 +- 3 files changed, 31 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ebe5a25..70b98aa 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,7 +3,7 @@ ## 0.5.0 - **First-use fetch no longer depends on GitHub's REST API quota.** Upstream docs are found with `git ls-remote`-style tag lookups and read from one streamed `codeload.github.com` tarball per repository, filtered to docs paths, instead of the REST tree and per-file requests (60 per hour per IP anonymously; 2 of 10 benchmark projects failed in our clean-quota run, up to 6 of 10 behind a shared IP). Cold-cache, no token: 10/10 on the 10-question version set, no 403, 1-5 GitHub connections per project instead of 2-28. `upstream::FORMAT` is 5. A missing tag error lists the tag names tried. An archive over 150 MB compressed for explicit fetch, or 64 MB for the automatic first-use fetch (or a codeload failure, or too little of the automatic 45 s budget left), falls back to the per-file REST path for that repository only, and the manifest records `via: codeload` or `rest`. Measured (cold cache, no token, 1 s pacing between GitHub connections, same 10-question set): before 8/10 correct, 2 projects hit 403, 152 GitHub connections (34 REST, 118 raw), 145 s total fetch time, 1.8 MB down; after 10/10, 0 403s, 31 connections (17 codeload, none REST or raw), 27 s, 56 MB down. Per package, next 14.2.35 and 15.1.0 download a 42-46 MB tarball in 2.7-5 s and TypeScript 5.6.3 a 32 MB one in 8 s, each over 2 connections, all through codeload. Explicit docs-site fetches also download whole-repository tarballs. `GITHUB_TOKEN`/`GH_TOKEN` stays on api.github.com for explicit fetch only, as in 0.4.0; git, codeload and raw requests carry no credential, so a private repository fails with "private repository: not supported". Caches of format 4 stay usable by automatic and offline queries (an explicit fetch still refreshes them). Trade-off: bytes go up (the tarball holds the whole repository) while connections and quota use go down. -- **Hybrid search is no longer worse than keyword search.** The fusion is now keyword-first: weighted reciprocal rank (90% keyword ranking with all docs boosts, 10% embedding ranking) replaces the min-max score mix that let the embedding outvote exact API-name matches, and the embedding is skipped when one entry leads by at least 2x. On the 105-question benchmark (package files, no upstream) hybrid goes from 60 to 64 of 105 against 59 for keyword-only, with the held-out set at 8/18 (was 6/18, keyword 8/18). +- **Hybrid search is no longer worse than keyword search.** The fusion is now keyword-first: on package files weighted reciprocal rank (90% keyword ranking with all docs boosts, 10% embedding ranking) replaces the min-max score mix that let the embedding outvote exact API-name matches; when upstream docs are indexed the normalized score mix is kept (rank-only fusion buried pages like tokio's select.md, 91 vs 96 of 105 fetched). The embedding is skipped when one entry leads by at least 2x. On the 105-question benchmark (package files, no upstream) hybrid goes from 60 to 64 of 105 against 59 for keyword-only, with the held-out set at 8/18 (was 6/18, keyword 8/18). - **Hybrid search is as light as keyword search.** A query reads only its own tokens' rows from the model file (positioned reads) instead of loading the 32 MB weight table: median CLI peak RSS 48 MB to 18 MB, median latency 42 ms to 18 ms (keyword-only 12 ms). The full model loads only when an index is built. - **CI floor.** The benchmark job fails when hybrid scores below keyword-only on the same docs (package-only or prefetched); the package-only hybrid floor rises from 60 to 62. diff --git a/crates/lockdocs-core/src/query.rs b/crates/lockdocs-core/src/query.rs index 5a58008..b3be774 100644 --- a/crates/lockdocs-core/src/query.rs +++ b/crates/lockdocs-core/src/query.rs @@ -23,6 +23,8 @@ const RELEVANCE_FLOOR: f32 = 0.3; const DENSE_WEIGHT: f32 = 0.1; /// Reciprocal-rank constant: `1 / (RRF_K + rank)`. const RRF_K: f32 = 10.0; +/// Share of the embedding in the score mix used when upstream docs are indexed. +const DOCS_DENSE_WEIGHT: f32 = 0.45; /// Embedding candidates considered per query. const DENSE_POOL: usize = 300; /// The keyword leader is final when it scores at least this many times the @@ -1120,6 +1122,7 @@ fn hybrid_rank( }; // Embedding rank of each entry among the 300 nearest to the query. let mut dense_rank: HashMap = HashMap::new(); + let mut dense_cos: HashMap = HashMap::new(); if let Some(qv) = &qv { let mut sims: Vec<(usize, f32)> = refs .par_iter() @@ -1130,6 +1133,11 @@ fn hybrid_rank( for (r, (i, _)) in sims.iter().take(DENSE_POOL).enumerate() { dense_rank.insert(*i, r); } + let cmax = sims.first().map_or(1.0, |s| s.1); + let floor = sims.get(300).map_or(0.0, |s| s.1); + for (i, c) in sims.iter().take(300) { + dense_cos.insert(*i, ((c - floor) / (cmax - floor).max(1e-6)).max(0.0)); + } if debug { for (i, c) in sims.iter().take(10) { let (pi, ei) = refs[*i]; @@ -1140,11 +1148,28 @@ fn hybrid_rank( if debug { eprintln!("decisive={decisive} dense={}", qv.is_some()); } - // Reciprocal-rank fusion, weighted towards the keyword ranking (the - // embedding breaks near-ties and rescues paraphrases, it cannot outvote - // an exact name or heading match). Scores are scaled so the best is 1. + // Package files only (API names and signatures): reciprocal-rank fusion + // weighted towards the keyword ranking, so the embedding breaks near-ties + // and rescues paraphrases but cannot outvote an exact name match. With + // upstream docs in the index the answers are prose pages, where the + // embedding is as reliable as keywords and score magnitudes matter (a + // rank-only fusion buried select.md under Graceful Shutdown): there each + // normalized score is mixed, as before. Scores are scaled so the best is 1. + let with_docs = ready.iter().any(|r| r.0.upstream.is_some()); let rrf = |r: usize| 1.0 / (RRF_K + r as f32); - let mut scored: Vec<(f32, usize, usize)> = if qv.is_some() { + let mut scored: Vec<(f32, usize, usize)> = if qv.is_some() && with_docs { + let mut ids: HashSet = lex.iter().map(|l| l.0).collect(); + ids.extend(dense_cos.keys().copied()); + let lexm: HashMap = fused.iter().map(|(&i, &(b, h))| (i, (1.0 - HEAD_WEIGHT) * b + HEAD_WEIGHT * h)).collect(); + ids.into_iter() + .map(|i| { + let (pi, ei) = refs[i]; + let l = lexm.get(&i).copied().unwrap_or(0.0); + let d = dense_cos.get(&i).copied().unwrap_or(0.0); + (((1.0 - DOCS_DENSE_WEIGHT) * l + DOCS_DENSE_WEIGHT * d) * weight(i), pi, ei) + }) + .collect() + } else if qv.is_some() { let mut ids: HashSet = lex.iter().map(|l| l.0).collect(); ids.extend(dense_rank.keys().copied()); let lex_rank: HashMap = lex.iter().enumerate().map(|(r, l)| (l.0, r)).collect(); diff --git a/docs/guide/how-it-works.md b/docs/guide/how-it-works.md index cd55e67..fff46da 100644 --- a/docs/guide/how-it-works.md +++ b/docs/guide/how-it-works.md @@ -4,7 +4,7 @@ 2. **Locate.** Each package's files are found on disk (see [Ecosystems](./ecosystems)). If the installed version differs from the lockfile, the answer says so and shows the installed one; with fetching enabled it reads the pinned version instead. 3. **Extract.** Markdown and reStructuredText are split into heading-scoped sections (long ones at paragraph boundaries, never inside code fences). HTML headings in MDX (`

@import

`) count as headings, and `export const title` names the page. Docs pages written as React components (Tailwind's installation guides) become sections too: step titles are headings, JSX text is prose, and `code:` strings are code blocks. Source files are parsed with tree-sitter (TypeScript, Python, Rust, Go) into symbols with a signature (the declaration without its body), a cleaned doc comment, a qualified path and a line number. 4. **Index.** Each entry becomes weighted terms: names and headings count most, doc prose next, signatures and code examples least. The tokenizer keeps identifiers whole and also splits camelCase, snake_case and digits; a light stemmer (plurals, -ing, -ed, -ation/-ate, -ability/-able) and a short programming synonym table (dict/dump, parse/validate, parameter/param, JavaScript/JS, ...) bridge wording. The index is cached on disk per package, version and source location, so a version is indexed once. -5. **Rank.** BM25 over the full text, BM25 over just the heading (or name) and first sentence, so a long body does not bury what an entry says it is, each normalized to its best hit. Then signals that matter for docs: entries whose name or heading contains a query word, identifiers written in the question (`model_dump`, `z.object`) and plain question words that name a documented top-level API of the package ("run code *after* the response" in Next.js, which exports `after`), changelog sections only when the question is about changes, and lower weight for private modules, bundled tooling, legacy copies, upgrade guides to an older major than yours ("Migrating to v6" in ESLint 9) and pages titled "(Deprecated)". The result is the keyword ranking. When one entry clearly leads (at least twice the runner-up) it is the answer and the embedding is not consulted. Otherwise an embedding ranking (a static model2vec model: a text's vector is the mean of its WordPiece token vectors) is fused in by weighted reciprocal rank, 90% keyword and 10% embedding: the embedding breaks near-ties and rescues paraphrases, but cannot outvote an exact API-name or heading match. A query is embedded by reading only its own tokens' rows from the model file, so the 32 MB weight table never enters memory. Generic headings such as Parameters, Returns or Examples take their topic from the heading above them. Deprecation notes among the top results ("use `model_validate` instead") lift the API they point to. +5. **Rank.** BM25 over the full text, BM25 over just the heading (or name) and first sentence, so a long body does not bury what an entry says it is, each normalized to its best hit. Then signals that matter for docs: entries whose name or heading contains a query word, identifiers written in the question (`model_dump`, `z.object`) and plain question words that name a documented top-level API of the package ("run code *after* the response" in Next.js, which exports `after`), changelog sections only when the question is about changes, and lower weight for private modules, bundled tooling, legacy copies, upgrade guides to an older major than yours ("Migrating to v6" in ESLint 9) and pages titled "(Deprecated)". The result is the keyword ranking. When one entry clearly leads (at least twice the runner-up) it is the answer and the embedding is not consulted. Otherwise (package files only) an embedding ranking (a static model2vec model: a text's vector is the mean of its WordPiece token vectors) is fused in by weighted reciprocal rank, 90% keyword and 10% embedding: the embedding breaks near-ties and rescues paraphrases, but cannot outvote an exact API-name or heading match. With upstream docs indexed, answers are prose pages and the normalized scores are mixed instead (45% embedding). A query is embedded by reading only its own tokens' rows from the model file, so the 32 MB weight table never enters memory. Generic headings such as Parameters, Returns or Examples take their topic from the heading above them. Deprecation notes among the top results ("use `model_validate` instead") lift the API they point to. 6. **Pack.** The top result also quotes the first paragraph of its page and parent section (with the short list or code block that follows), when they add something: a deprecation notice at the top of the page, or the setup a subsection builds on. Results are added in rank order until the token budget is spent (the last one trimmed at a line boundary), each with a `package@version path:line` citation. `api` is exact rather than ranked: it finds entries whose name matches the last segment of the symbol, scores qualifiers (`Router` in `axum::Router::route`), follows re-exports, and adds overloads, members and other matches. From b1822cb373cdd4e4f46dc2721458515cc9e949a7 Mon Sep 17 00:00:00 2001 From: Kyle Tse Date: Fri, 2 Oct 2026 13:14:10 +0000 Subject: [PATCH 3/4] fix: honour LOCKDOCS_HEAD_WEIGHT in the docs-mode mix; benchmarks page lists five configurations --- crates/lockdocs-core/src/query.rs | 2 +- docs/benchmarks.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/crates/lockdocs-core/src/query.rs b/crates/lockdocs-core/src/query.rs index b3be774..f8269bb 100644 --- a/crates/lockdocs-core/src/query.rs +++ b/crates/lockdocs-core/src/query.rs @@ -1160,7 +1160,7 @@ fn hybrid_rank( let mut scored: Vec<(f32, usize, usize)> = if qv.is_some() && with_docs { let mut ids: HashSet = lex.iter().map(|l| l.0).collect(); ids.extend(dense_cos.keys().copied()); - let lexm: HashMap = fused.iter().map(|(&i, &(b, h))| (i, (1.0 - HEAD_WEIGHT) * b + HEAD_WEIGHT * h)).collect(); + let lexm: HashMap = fused.iter().map(|(&i, &(b, h))| (i, (1.0 - head_w) * b + head_w * h)).collect(); ids.into_iter() .map(|i| { let (pi, ei) = refs[i]; diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 7d7daf0..a718592 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -6,7 +6,7 @@ - **Held-out questions:** 18 questions marked `held-out` are not used for tuning: they were written before the ranking changes they measure and have their own column. Held-out questions that are later used to diagnose a miss join the main set (their `history` field says so), and new ones replace them. - **Projects:** one project per version in [`bench/projects`](https://github.com/SylphxAI/lockdocs/tree/main/bench/projects), installed at those pins by [`bench/setup.sh`](https://github.com/SylphxAI/lockdocs/blob/main/bench/setup.sh). - **Grading:** an answer passes when it contains at least one string from every `expect` group (the version-correct API, in code or prose form) and none of the `reject` strings (the other version's API). Case-insensitive; the same grader for every tool. -- **lockdocs**, default settings (1,200-token budget), in four configurations: keyword only (`LOCKDOCS_EMBED=0`) on package files; hybrid on package files (`LOCKDOCS_FETCH=0`, upstream disabled); real first-use defaults in an initially empty isolated `LOCKDOCS_CACHE` with fetch-policy and credential variables removed, before explicit prefetch; hybrid after explicit `lockdocs fetch` added upstream docs and major-version sites. The first-use row includes cold-download latency and anonymous rate-limit failures; it is never copied from the prefetched score. Full runs enforce at least 96/105 prefetched and 62/105 package-only hybrid, and fail if hybrid scores below keyword-only on the same docs (package-only or prefetched). Index and fetch times are reported separately. +- **lockdocs**, default settings (1,200-token budget), in five configurations: keyword only (`LOCKDOCS_EMBED=0`) on package files; hybrid on package files (`LOCKDOCS_FETCH=0`, upstream disabled); real first-use defaults in an initially empty isolated `LOCKDOCS_CACHE` with fetch-policy and credential variables removed, before explicit prefetch; hybrid after explicit `lockdocs fetch` added upstream docs and major-version sites; and keyword only on that same fetched index (fetched-keyword), the baseline the fetched hybrid must beat. The first-use row includes cold-download latency and anonymous rate-limit failures; it is never copied from the prefetched score. Full runs enforce at least 96/105 prefetched and 62/105 package-only hybrid, and fail if hybrid scores below keyword-only on the same docs (package-only or prefetched). Index and fetch times are reported separately. - **Context7:** the anonymous API as its MCP server uses it: search the library, pick the top result and its listed version with the same major (exact when listed), then fetch context for the question. When that library answers HTTP 404, the next search result is tried, as an agent would. Rate-limit responses are recorded, not retried. - **Tokens:** tiktoken `o200k_base`. **Latency:** wall time per call from the same GitHub-hosted runner (for lockdocs: a fresh CLI process per question, including loading the model). - **Runner:** [`bench/run.py`](https://github.com/SylphxAI/lockdocs/blob/main/bench/run.py) via the [`bench` workflow](https://github.com/SylphxAI/lockdocs/actions/workflows/bench.yml). Reproduce: `bash bench/setup.sh && python3 bench/run.py target/release/lockdocs bench/projects out.json --fetch --context7`. From 72ab9647ee06ec4e72c1251d131b08d742504d83 Mon Sep 17 00:00:00 2001 From: Kyle Tse Date: Fri, 2 Oct 2026 13:30:09 +0000 Subject: [PATCH 4/4] docs(changelog): hybrid fusion entries go under Unreleased, after 0.5.0 --- CHANGELOG.md | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 70b98aa..9d445e6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,12 +1,15 @@ # Changelog -## 0.5.0 +## Unreleased -- **First-use fetch no longer depends on GitHub's REST API quota.** Upstream docs are found with `git ls-remote`-style tag lookups and read from one streamed `codeload.github.com` tarball per repository, filtered to docs paths, instead of the REST tree and per-file requests (60 per hour per IP anonymously; 2 of 10 benchmark projects failed in our clean-quota run, up to 6 of 10 behind a shared IP). Cold-cache, no token: 10/10 on the 10-question version set, no 403, 1-5 GitHub connections per project instead of 2-28. `upstream::FORMAT` is 5. A missing tag error lists the tag names tried. An archive over 150 MB compressed for explicit fetch, or 64 MB for the automatic first-use fetch (or a codeload failure, or too little of the automatic 45 s budget left), falls back to the per-file REST path for that repository only, and the manifest records `via: codeload` or `rest`. Measured (cold cache, no token, 1 s pacing between GitHub connections, same 10-question set): before 8/10 correct, 2 projects hit 403, 152 GitHub connections (34 REST, 118 raw), 145 s total fetch time, 1.8 MB down; after 10/10, 0 403s, 31 connections (17 codeload, none REST or raw), 27 s, 56 MB down. Per package, next 14.2.35 and 15.1.0 download a 42-46 MB tarball in 2.7-5 s and TypeScript 5.6.3 a 32 MB one in 8 s, each over 2 connections, all through codeload. Explicit docs-site fetches also download whole-repository tarballs. `GITHUB_TOKEN`/`GH_TOKEN` stays on api.github.com for explicit fetch only, as in 0.4.0; git, codeload and raw requests carry no credential, so a private repository fails with "private repository: not supported". Caches of format 4 stay usable by automatic and offline queries (an explicit fetch still refreshes them). Trade-off: bytes go up (the tarball holds the whole repository) while connections and quota use go down. - **Hybrid search is no longer worse than keyword search.** The fusion is now keyword-first: on package files weighted reciprocal rank (90% keyword ranking with all docs boosts, 10% embedding ranking) replaces the min-max score mix that let the embedding outvote exact API-name matches; when upstream docs are indexed the normalized score mix is kept (rank-only fusion buried pages like tokio's select.md, 91 vs 96 of 105 fetched). The embedding is skipped when one entry leads by at least 2x. On the 105-question benchmark (package files, no upstream) hybrid goes from 60 to 64 of 105 against 59 for keyword-only, with the held-out set at 8/18 (was 6/18, keyword 8/18). - **Hybrid search is as light as keyword search.** A query reads only its own tokens' rows from the model file (positioned reads) instead of loading the 32 MB weight table: median CLI peak RSS 48 MB to 18 MB, median latency 42 ms to 18 ms (keyword-only 12 ms). The full model loads only when an index is built. - **CI floor.** The benchmark job fails when hybrid scores below keyword-only on the same docs (package-only or prefetched); the package-only hybrid floor rises from 60 to 62. +## 0.5.0 + +- **First-use fetch no longer depends on GitHub's REST API quota.** Upstream docs are found with `git ls-remote`-style tag lookups and read from one streamed `codeload.github.com` tarball per repository, filtered to docs paths, instead of the REST tree and per-file requests (60 per hour per IP anonymously; 2 of 10 benchmark projects failed in our clean-quota run, up to 6 of 10 behind a shared IP). Cold-cache, no token: 10/10 on the 10-question version set, no 403, 1-5 GitHub connections per project instead of 2-28. `upstream::FORMAT` is 5. A missing tag error lists the tag names tried. An archive over 150 MB compressed for explicit fetch, or 64 MB for the automatic first-use fetch (or a codeload failure, or too little of the automatic 45 s budget left), falls back to the per-file REST path for that repository only, and the manifest records `via: codeload` or `rest`. Measured (cold cache, no token, 1 s pacing between GitHub connections, same 10-question set): before 8/10 correct, 2 projects hit 403, 152 GitHub connections (34 REST, 118 raw), 145 s total fetch time, 1.8 MB down; after 10/10, 0 403s, 31 connections (17 codeload, none REST or raw), 27 s, 56 MB down. Per package, next 14.2.35 and 15.1.0 download a 42-46 MB tarball in 2.7-5 s and TypeScript 5.6.3 a 32 MB one in 8 s, each over 2 connections, all through codeload. Explicit docs-site fetches also download whole-repository tarballs. `GITHUB_TOKEN`/`GH_TOKEN` stays on api.github.com for explicit fetch only, as in 0.4.0; git, codeload and raw requests carry no credential, so a private repository fails with "private repository: not supported". Caches of format 4 stay usable by automatic and offline queries (an explicit fetch still refreshes them). Trade-off: bytes go up (the tarball holds the whole repository) while connections and quota use go down. + ## 0.4.0 - **Useful docs on first use.** `docs` and `api` now automatically add public upstream docs for the requested release, anonymously. The existing fetcher resolves exact `refs/tags/` refs, rejects same-named branches, and peels annotated tags to an immutable commit before downloading bounded docs. Missing registry packages and major-version website docs remain explicit opt-in.