From 2e7cb2d2bc720fd3b709726882f2f813c8a3cce2 Mon Sep 17 00:00:00 2001 From: Kyle Tse Date: Wed, 30 Sep 2026 23:50:23 +0000 Subject: [PATCH] refactor: adopt mcp-kit 0.3 and remove shared copies --- Cargo.lock | 83 ++++--- PROJECT.md | 3 +- crates/lockdocs-core/Cargo.toml | 2 +- crates/lockdocs-core/src/cache.rs | 5 +- crates/lockdocs-core/src/embed.rs | 358 --------------------------- crates/lockdocs-core/src/index.rs | 2 +- crates/lockdocs-core/src/lib.rs | 6 +- crates/lockdocs-core/src/semantic.rs | 78 ++++++ crates/lockdocs-core/src/tokenize.rs | 138 ----------- crates/lockdocs/Cargo.toml | 2 +- crates/lockdocs/src/main.rs | 8 +- crates/lockdocs/src/star_hint.rs | 134 ---------- docs/capabilities.md | 2 +- docs/guide/how-it-works.md | 7 + 14 files changed, 154 insertions(+), 674 deletions(-) delete mode 100644 crates/lockdocs-core/src/embed.rs create mode 100644 crates/lockdocs-core/src/semantic.rs delete mode 100644 crates/lockdocs-core/src/tokenize.rs delete mode 100644 crates/lockdocs/src/star_hint.rs diff --git a/Cargo.lock b/Cargo.lock index bc24ad8..720cc69 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -67,11 +67,11 @@ checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" [[package]] name = "block-buffer" -version = "0.12.1" +version = "0.10.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" dependencies = [ - "hybrid-array", + "generic-array", ] [[package]] @@ -129,12 +129,6 @@ dependencies = [ "thiserror", ] -[[package]] -name = "const-oid" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a6ef517f0926dd24a1582492c791b6a4818a4d94e789a334894aa15b0d12f55c" - [[package]] name = "core-foundation-sys" version = "0.8.7" @@ -143,9 +137,9 @@ checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" [[package]] name = "cpufeatures" -version = "0.3.1" +version = "0.2.17" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" dependencies = [ "libc", ] @@ -192,21 +186,21 @@ checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6" [[package]] name = "crypto-common" -version = "0.2.2" +version = "0.1.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce6e4c961d6cd6c9a86db418387425e8bdeaf05b3c8bc1411e6dca4c252f1453" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" dependencies = [ - "hybrid-array", + "generic-array", + "typenum", ] [[package]] name = "digest" -version = "0.11.3" +version = "0.10.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" dependencies = [ "block-buffer", - "const-oid", "crypto-common", ] @@ -416,6 +410,16 @@ dependencies = [ "slab", ] +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + [[package]] name = "getrandom" version = "0.2.17" @@ -495,15 +499,6 @@ version = "1.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" -[[package]] -name = "hybrid-array" -version = "0.4.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "27f864f10dfb56725ce5ce5472bc52252c8f93a4ab86327122cebf62c5f59a17" -dependencies = [ - "typenum", -] - [[package]] name = "iana-time-zone" version = "0.1.65" @@ -608,7 +603,7 @@ dependencies = [ "rayon", "serde", "serde_json", - "sha2", + "sylphx-mcp-kit", "tar", "tempfile", "toml", @@ -1001,9 +996,9 @@ dependencies = [ [[package]] name = "sha2" -version = "0.11.0" +version = "0.10.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "446ba717509524cb3f22f17ecc096f10f4822d76ab5c0b9822c5f9c284e825f4" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" dependencies = [ "cfg-if", "cpufeatures", @@ -1057,15 +1052,20 @@ checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" [[package]] name = "sylphx-mcp-kit" -version = "0.1.0" -source = "git+https://github.com/SylphxAI/mcp-kit?tag=v0.1.0#67a70b4fffc0f0dbb7a5dafb3baf1b0d1a9afab4" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "baba92c4baf8d691a65fbd0776db27a91e20e4d9f032c7845d1788b8eae7ac58" dependencies = [ "anyhow", "dirs 6.0.0", "rmcp", + "serde", "serde_json", + "sha2", "tokio", "toml_edit", + "unicode-normalization", + "ureq", ] [[package]] @@ -1134,6 +1134,12 @@ dependencies = [ "syn 3.0.6", ] +[[package]] +name = "tinyvec" +version = "1.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fd3ca314f692efd6c868f8408f53fe444634a845f96c028b97d35f6a1f79f0ee" + [[package]] name = "tokio" version = "1.53.1" @@ -1339,6 +1345,15 @@ version = "1.0.26" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" +[[package]] +name = "unicode-normalization" +version = "0.1.25" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8" +dependencies = [ + "tinyvec", +] + [[package]] name = "untrusted" version = "0.9.0" @@ -1391,6 +1406,12 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + [[package]] name = "wasi" version = "0.11.1+wasi-snapshot-preview1" diff --git a/PROJECT.md b/PROJECT.md index 7d60d73..ee99656 100644 --- a/PROJECT.md +++ b/PROJECT.md @@ -15,7 +15,8 @@ MIT licensed. fetch (`fetch.rs`), Markdown sections (`markdown.rs`), tree-sitter symbol extraction for TypeScript, Python, Rust and Go (`extract.rs`), BM25 (`bm25.rs`), the per-version index cache (`index.rs`), and the three queries - (`query.rs`) + (`query.rs`), and embedding policy (`semantic.rs`, with the engine and + tokenizer from `sylphx-mcp-kit` 0.3) - `crates/lockdocs`: the `lockdocs` binary (CLI, MCP stdio server, `setup`) - `packages/lockdocs`: the npm launcher; `packages/npm/*`: native binaries - `bench/`: version-sensitive questions, project pins, and the runner diff --git a/crates/lockdocs-core/Cargo.toml b/crates/lockdocs-core/Cargo.toml index 67302a5..2df3f70 100644 --- a/crates/lockdocs-core/Cargo.toml +++ b/crates/lockdocs-core/Cargo.toml @@ -24,7 +24,7 @@ tree-sitter-rust = "0.24" ureq = "3" flate2 = "1" tar = "0.4" -sha2 = "0.11" +sylphx-mcp-kit = { version = "0.3", default-features = false, features = ["embed", "search"] } zip = { version = "8", default-features = false, features = ["deflate-flate2-zlib-rs"] } [dev-dependencies] diff --git a/crates/lockdocs-core/src/cache.rs b/crates/lockdocs-core/src/cache.rs index 88d941b..96e979e 100644 --- a/crates/lockdocs-core/src/cache.rs +++ b/crates/lockdocs-core/src/cache.rs @@ -4,10 +4,7 @@ use std::path::PathBuf; /// `LOCKDOCS_CACHE`, else the OS cache directory, else a temp dir. pub fn dir() -> PathBuf { - if let Some(p) = std::env::var_os("LOCKDOCS_CACHE").filter(|p| !p.is_empty()) { - return PathBuf::from(p); - } - dirs::cache_dir().unwrap_or_else(std::env::temp_dir).join("lockdocs") + mcp_kit::cache::root("LOCKDOCS_CACHE", "lockdocs", mcp_kit::cache::Fallback::Temp, mcp_kit::cache::Override::NonEmpty).expect("temporary cache fallback") } /// FNV-1a 64: stable across builds and platforms. diff --git a/crates/lockdocs-core/src/embed.rs b/crates/lockdocs-core/src/embed.rs deleted file mode 100644 index 610bcfa..0000000 --- a/crates/lockdocs-core/src/embed.rs +++ /dev/null @@ -1,358 +0,0 @@ -//! Dense retrieval with a small static embedding model (model2vec -//! potion-retrieval-32M, distilled from bge-base-en-v1.5, MIT). A text's -//! embedding is the normalized mean of its WordPiece tokens' vectors, so -//! embedding every symbol of a large package takes milliseconds on a CPU. -//! -//! The model (129 MB) is downloaded once, verified by SHA-256, quantized to -//! int8 (32 MB) in the cache, and used offline afterwards. Without it, -//! lockdocs is BM25-only. - -use anyhow::{bail, Context, Result}; -use sha2::{Digest, Sha256}; -use std::collections::HashMap; -use std::io::{Read, Write}; -use std::path::PathBuf; -use std::sync::{Arc, OnceLock, RwLock}; - -pub const MODEL_ID: &str = "potion-retrieval-32M"; -const REPO: &str = "minishlab/potion-retrieval-32M"; -const REVISION: &str = "6fc8051fab2a1e0ee76689cf08c853792ac285e7"; -const WEIGHTS_SHA256: &str = "07609e5bd33aad37900b3fd62f4ec96f6daec88ca4d46b9d8b928bfababf6ea0"; -const WEIGHTS_BYTES: u64 = 129_210_456; -const MAX_WORD_CHARS: usize = 100; - -pub struct Model { - vocab: HashMap, - dims: usize, - /// Row-major int8 weights (stored as bytes) and one scale per row. - q: Vec, - scale: Vec, -} - -/// An embedding: int8 values and a scale (value = q * scale); unit length. -#[derive(Clone, Debug, serde::Serialize, serde::Deserialize, Default)] -pub struct Vec8 { - pub q: Vec, - pub s: f32, -} - -fn model_dir() -> PathBuf { - crate::cache::dir().join("models").join(MODEL_ID) -} - -/// Embeddings are on unless `LOCKDOCS_EMBED=0`. -pub fn enabled() -> bool { - !matches!(std::env::var("LOCKDOCS_EMBED").as_deref(), Ok("0") | Ok("false") | Ok("off")) -} - -/// Is the model downloaded and converted? -pub fn installed() -> bool { - let d = model_dir(); - d.join("model.q8").is_file() && d.join("vocab.txt").is_file() -} - -static MODEL: OnceLock>>> = OnceLock::new(); - -/// The model when installed and enabled (loaded once per process). -pub fn get() -> Option> { - if !enabled() { - return None; - } - let cell = MODEL.get_or_init(|| RwLock::new(None)); - if let Some(m) = cell.read().unwrap().as_ref() { - return Some(m.clone()); - } - if !installed() { - return None; - } - let m = Arc::new(Model::load().ok()?); - *cell.write().unwrap() = Some(m.clone()); - Some(m) -} - -/// Download, verify and convert the model if it is missing. Prints one line -/// to stderr before downloading. -pub fn ensure() -> Result<()> { - if installed() { - return Ok(()); - } - let dir = model_dir(); - std::fs::create_dir_all(&dir)?; - eprintln!( - "lockdocs: downloading the embedding model {MODEL_ID} ({} MB, once) from huggingface.co to {}. Set LOCKDOCS_EMBED=0 to stay keyword-only.", - WEIGHTS_BYTES / 1_000_000, - crate::locate::tilde(&dir) - ); - let base = std::env::var("LOCKDOCS_MODEL_URL").unwrap_or_else(|_| format!("https://huggingface.co/{REPO}/resolve/{REVISION}")); - let agent: ureq::Agent = ureq::Agent::config_builder() - .timeout_global(Some(std::time::Duration::from_secs(900))) - .user_agent(concat!("lockdocs/", env!("CARGO_PKG_VERSION"))) - .build() - .into(); - // Tokenizer vocab - let mut tok = String::new(); - agent - .get(&format!("{base}/tokenizer.json")) - .call() - .context("downloading tokenizer.json")? - .body_mut() - .as_reader() - .take(16 << 20) - .read_to_string(&mut tok)?; - let t: serde_json::Value = serde_json::from_str(&tok)?; - let vocab = t - .pointer("/model/vocab") - .and_then(|v| v.as_object()) - .context("tokenizer.json has no WordPiece vocab")?; - let mut rows: Vec<(&String, u64)> = vocab.iter().filter_map(|(k, v)| v.as_u64().map(|i| (k, i))).collect(); - rows.sort_by_key(|r| r.1); - // Weights, hashed while streaming. - let tmp = dir.join("model.safetensors.part"); - let mut res = agent - .get(&format!("{base}/model.safetensors")) - .call() - .context("downloading model.safetensors")?; - let mut reader = res.body_mut().as_reader().take(WEIGHTS_BYTES + 1); - let mut file = std::fs::File::create(&tmp)?; - let mut hasher = Sha256::new(); - let mut buf = vec![0u8; 1 << 20]; - let mut total = 0u64; - loop { - let n = reader.read(&mut buf)?; - if n == 0 { - break; - } - hasher.update(&buf[..n]); - file.write_all(&buf[..n])?; - total += n as u64; - } - drop(file); - let got: String = hasher.finalize().iter().map(|b| format!("{b:02x}")).collect(); - if got != WEIGHTS_SHA256 || total != WEIGHTS_BYTES { - let _ = std::fs::remove_file(&tmp); - bail!("model download failed verification (sha256 {got}, {total} bytes)"); - } - let bytes = std::fs::read(&tmp)?; - let (dims, data) = parse_safetensors(&bytes)?; - if data.len() / dims != rows.len() { - bail!("model has {} rows but the vocab has {}", data.len() / dims, rows.len()); - } - // Quantize per row to int8. - let n = rows.len(); - let mut out = Vec::with_capacity(8 + n * 4 + n * dims); - out.extend_from_slice(&(n as u32).to_le_bytes()); - out.extend_from_slice(&(dims as u32).to_le_bytes()); - let mut q = Vec::with_capacity(n * dims); - for r in 0..n { - let row = &data[r * dims..(r + 1) * dims]; - let max = row.iter().fold(0f32, |m, x| m.max(x.abs())).max(1e-12); - let s = max / 127.0; - out.extend_from_slice(&s.to_le_bytes()); - q.extend(row.iter().map(|x| (x / s).round().clamp(-127.0, 127.0) as i8 as u8)); - } - out.extend_from_slice(&q); - std::fs::write(dir.join("model.q8.part"), &out)?; - let words: Vec<&str> = rows.iter().map(|r| r.0.as_str()).collect(); - std::fs::write(dir.join("vocab.txt"), words.join("\n"))?; - std::fs::rename(dir.join("model.q8.part"), dir.join("model.q8"))?; - let _ = std::fs::remove_file(&tmp); - Ok(()) -} - -fn parse_safetensors(bytes: &[u8]) -> Result<(usize, Vec)> { - let hlen = u64::from_le_bytes(bytes.get(..8).context("short file")?.try_into()?) as usize; - let header: serde_json::Value = serde_json::from_slice(bytes.get(8..8 + hlen).context("short header")?)?; - let t = header.get("embeddings").context("no embeddings tensor")?; - if t.get("dtype").and_then(|d| d.as_str()) != Some("F32") { - bail!("unexpected dtype"); - } - let dims = t.pointer("/shape/1").and_then(|d| d.as_u64()).context("no shape")? as usize; - let start = 8 + hlen + t.pointer("/data_offsets/0").and_then(|d| d.as_u64()).context("offsets")? as usize; - let end = 8 + hlen + t.pointer("/data_offsets/1").and_then(|d| d.as_u64()).context("offsets")? as usize; - let raw = bytes.get(start..end).context("short data")?; - Ok((dims, raw.as_chunks::<4>().0.iter().map(|c| f32::from_le_bytes(*c)).collect())) -} - -impl Model { - fn load() -> Result { - let d = model_dir(); - let mut bytes = std::fs::read(d.join("model.q8"))?; - let n = u32::from_le_bytes(bytes[0..4].try_into()?) as usize; - let dims = u32::from_le_bytes(bytes[4..8].try_into()?) as usize; - if bytes.len() != 8 + n * 4 + n * dims { - bail!("model.q8 is truncated"); - } - let scale: Vec = bytes[8..8 + n * 4].as_chunks::<4>().0.iter().map(|c| f32::from_le_bytes(*c)).collect(); - let q = bytes.split_off(8 + n * 4); - let vocab: HashMap = std::fs::read_to_string(d.join("vocab.txt"))? - .split('\n') - .enumerate() - .map(|(i, w)| (w.to_string(), i as u32)) - .collect(); - Ok(Model { vocab, dims, q, scale }) - } - - pub fn dims(&self) -> usize { - self.dims - } - - /// WordPiece ids for a text (BERT uncased normalization; identifiers are - /// split on camelCase and snake_case first). - pub fn tokenize(&self, text: &str) -> Vec { - let mut ids = Vec::new(); - for word in pre_tokenize(text) { - if word.chars().count() > MAX_WORD_CHARS { - continue; - } - let chars: Vec = word.chars().collect(); - let mut start = 0; - let mut pieces = Vec::new(); - let mut ok = true; - while start < chars.len() { - let mut end = chars.len(); - let mut found = None; - while start < end { - let mut s: String = chars[start..end].iter().collect(); - if start > 0 { - s.insert_str(0, "##"); - } - if let Some(id) = self.vocab.get(&s) { - found = Some(*id); - break; - } - end -= 1; - } - match found { - Some(id) => { - pieces.push(id); - start = end; - } - None => { - ok = false; - break; - } - } - } - if ok { - ids.extend(pieces); - } - } - ids - } - - /// Unit-length embedding of a text, or None when no token is known. - pub fn embed(&self, text: &str) -> Option> { - let ids = self.tokenize(text); - if ids.is_empty() { - return None; - } - let mut acc = vec![0f32; self.dims]; - for id in &ids { - let r = *id as usize; - let s = self.scale[r]; - for (a, q) in acc.iter_mut().zip(&self.q[r * self.dims..(r + 1) * self.dims]) { - *a += *q as i8 as f32 * s; - } - } - let norm = acc.iter().map(|x| x * x).sum::().sqrt(); - if norm < 1e-9 { - return None; - } - acc.iter_mut().for_each(|x| *x /= norm); - Some(acc) - } - - pub fn embed8(&self, text: &str) -> Vec8 { - match self.embed(text) { - Some(v) => quantize(&v), - None => Vec8::default(), - } - } -} - -pub fn quantize(v: &[f32]) -> Vec8 { - let max = v.iter().fold(0f32, |m, x| m.max(x.abs())).max(1e-12); - let s = max / 127.0; - Vec8 { - q: v.iter().map(|x| (x / s).round().clamp(-127.0, 127.0) as i8).collect(), - s, - } -} - -/// Cosine similarity of a unit query with a stored unit vector. -pub fn cosine(q: &[f32], v: &Vec8) -> f32 { - if v.q.len() != q.len() { - return 0.0; - } - let mut dot = 0f32; - for (a, b) in q.iter().zip(&v.q) { - dot += a * *b as f32; - } - dot * v.s -} - -/// BERT basic tokenization with identifier splitting: lowercase words and -/// single punctuation characters. -fn pre_tokenize(text: &str) -> Vec { - let mut spaced = String::with_capacity(text.len() + 16); - let mut prev: Option = None; - for c in text.chars() { - if let Some(p) = prev { - // camelCase / PascalCase boundary - if c.is_uppercase() && p.is_lowercase() { - spaced.push(' '); - } - } - spaced.push(if c == '_' { ' ' } else { c }); - prev = Some(c); - } - let mut out = Vec::new(); - let mut cur = String::new(); - for c in spaced.chars() { - if c.is_whitespace() || c.is_control() { - if !cur.is_empty() { - out.push(std::mem::take(&mut cur)); - } - } else if c.is_ascii_punctuation() || (!c.is_alphanumeric() && !c.is_whitespace()) { - if !cur.is_empty() { - out.push(std::mem::take(&mut cur)); - } - out.push(c.to_string()); - } else { - cur.extend(c.to_lowercase()); - } - } - if !cur.is_empty() { - out.push(cur); - } - out -} - -#[cfg(test)] -mod tests { - use super::*; - #[test] - fn wordpiece_and_embedding() { - let words = ["[PAD]", "[UNK]", "model", "valid", "##ate", "dict", "strict", "object", "(", ")"]; - let vocab: HashMap = words.iter().enumerate().map(|(i, w)| (w.to_string(), i as u32)).collect(); - let dims = 2; - let mut q = Vec::new(); - for i in 0..words.len() { - q.push(((i % 3) * 10) as u8); - q.push((((i + 1) % 2) * 10) as u8); - } - let m = Model { - vocab, - dims, - q, - scale: vec![0.1; words.len()], - }; - assert_eq!(m.tokenize("model_validate(dict)"), vec![2, 3, 4, 8, 5, 9]); - assert_eq!(m.tokenize("strictObject"), vec![6, 7]); - let e = m.embed("model dict").unwrap(); - assert!((e.iter().map(|x| x * x).sum::() - 1.0).abs() < 1e-4); - let v = quantize(&e); - assert!((cosine(&e, &v) - 1.0).abs() < 0.02); - assert!(m.embed("zzzz").is_none()); - } -} diff --git a/crates/lockdocs-core/src/index.rs b/crates/lockdocs-core/src/index.rs index 46f363c..dcf926e 100644 --- a/crates/lockdocs-core/src/index.rs +++ b/crates/lockdocs-core/src/index.rs @@ -193,7 +193,7 @@ pub fn build(dep: &Dep, src: &Source, root: &Path, up: Option<&(PathBuf, Manifes let vecs: Vec = match &model { Some(m) => { use rayon::prelude::*; - entries.par_iter().map(|e| m.embed8(&embed_text(e))).collect() + entries.par_iter().map(|e| m.embed8(&embed_text(e)).unwrap_or_default()).collect() } None => Vec::new(), }; diff --git a/crates/lockdocs-core/src/lib.rs b/crates/lockdocs-core/src/lib.rs index 24898e1..70e3883 100644 --- a/crates/lockdocs-core/src/lib.rs +++ b/crates/lockdocs-core/src/lib.rs @@ -4,7 +4,9 @@ pub mod bm25; pub mod cache; -pub mod embed; +pub mod semantic; +// Preserve the public module names without retaining local implementations. +pub use semantic as embed; pub mod extract; pub mod fetch; pub mod index; @@ -13,7 +15,7 @@ pub mod lockfile; pub mod markdown; pub mod project; pub mod query; -pub mod tokenize; +pub use mcp_kit::search as tokenize; pub mod upstream; use serde::{Deserialize, Serialize}; diff --git a/crates/lockdocs-core/src/semantic.rs b/crates/lockdocs-core/src/semantic.rs new file mode 100644 index 0000000..38c67be --- /dev/null +++ b/crates/lockdocs-core/src/semantic.rs @@ -0,0 +1,78 @@ +//! Product policy for the shared embedding engine: the retrieval model, +//! identifier-aware inputs, lockdocs' opt-out and existing cache location. + +use anyhow::Result; +use mcp_kit::embed::{self, Model, Tokenization, POTION_RETRIEVAL_32M}; +use std::path::PathBuf; +use std::sync::{Arc, OnceLock, RwLock}; + +pub use mcp_kit::embed::{cosine, quantize, Vec8}; +pub const MODEL_ID: &str = POTION_RETRIEVAL_32M.id; + +fn model_dir() -> PathBuf { + crate::cache::dir().join("models").join(MODEL_ID) +} + +/// Embeddings are on unless `LOCKDOCS_EMBED=0` (or false/off). +pub fn enabled() -> bool { + !matches!(std::env::var("LOCKDOCS_EMBED").as_deref(), Ok("0") | Ok("false") | Ok("off")) +} + +pub fn installed() -> bool { + embed::installed_at(&model_dir()) +} + +static MODEL: OnceLock>>> = OnceLock::new(); + +/// Load existing model.q8/vocab.txt directly. The shared input policy keeps +/// cached entry vectors and new query vectors in the same embedding space. +pub fn get() -> Option> { + if !enabled() { + return None; + } + let cell = MODEL.get_or_init(|| RwLock::new(None)); + if let Some(model) = cell.read().unwrap().as_ref() { + return Some(model.clone()); + } + if !installed() { + return None; + } + let model = Arc::new(Model::load_dir(&model_dir()).ok()?.with_tokenization(Tokenization::Identifiers)); + *cell.write().unwrap() = Some(model.clone()); + Some(model) +} + +pub fn ensure() -> Result<()> { + let base = std::env::var("LOCKDOCS_MODEL_URL") + .unwrap_or_else(|_| format!("https://huggingface.co/{}/resolve/{}", POTION_RETRIEVAL_32M.repo, POTION_RETRIEVAL_32M.revision)); + embed::ensure_at( + &POTION_RETRIEVAL_32M, + &model_dir(), + "lockdocs", + "Set LOCKDOCS_EMBED=0 to stay keyword-only.", + Some(&base), + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn persisted_entry_vectors_keep_their_postcard_layout() { + #[derive(serde::Serialize)] + struct PreviousVec8 { + q: Vec, + s: f32, + } + let previous = PreviousVec8 { + q: vec![-127, 0, 127], + s: 0.01, + }; + let bytes = postcard::to_stdvec(&previous).unwrap(); + let current: Vec8 = postcard::from_bytes(&bytes).unwrap(); + assert_eq!(current.q, previous.q); + assert_eq!(current.s, previous.s); + assert_eq!(postcard::to_stdvec(¤t).unwrap(), bytes); + } +} diff --git a/crates/lockdocs-core/src/tokenize.rs b/crates/lockdocs-core/src/tokenize.rs deleted file mode 100644 index 6fe8731..0000000 --- a/crates/lockdocs-core/src/tokenize.rs +++ /dev/null @@ -1,138 +0,0 @@ -//! Code-aware tokenizer: identifiers are kept whole and also split on -//! camelCase, PascalCase, snake_case and digits, all lowercased. - -use std::collections::HashMap; - -/// Push the whole identifier and its sub-words. -fn push_ident(word: &str, out: &mut Vec) { - if word.len() < 2 || word.len() > 64 { - return; - } - let lower = word.to_ascii_lowercase(); - let mut parts: Vec = Vec::new(); - for piece in word.split('_').filter(|p| !p.is_empty()) { - let chars: Vec = piece.chars().collect(); - let mut cur = String::new(); - for i in 0..chars.len() { - let c = chars[i]; - let boundary = i > 0 - && ((c.is_ascii_uppercase() - && (chars[i - 1].is_ascii_lowercase() - || chars[i - 1].is_ascii_digit() - || (i + 1 < chars.len() && chars[i + 1].is_ascii_lowercase() && chars[i - 1].is_ascii_uppercase()))) - || (c.is_ascii_digit() != chars[i - 1].is_ascii_digit())); - if boundary && !cur.is_empty() { - parts.push(std::mem::take(&mut cur)); - } - cur.push(c.to_ascii_lowercase()); - } - if !cur.is_empty() { - parts.push(cur); - } - } - let multi = parts.len() > 1; - out.push(lower); - if multi { - for p in parts { - if p.len() >= 2 && !p.chars().all(|c| c.is_ascii_digit()) { - out.push(p); - } - } - } -} - -/// Tokenize free text or code. -pub fn tokenize(text: &str) -> Vec { - let mut out = Vec::new(); - let mut start: Option = None; - for (i, ch) in text.char_indices() { - let word = ch.is_ascii_alphanumeric() || ch == '_'; - match (word, start) { - (true, None) => start = Some(i), - (false, Some(s)) => { - push_ident(&text[s..i], &mut out); - start = None; - } - _ => {} - } - } - if let Some(s) = start { - push_ident(&text[s..], &mut out); - } - out -} - -/// Terms contributed by a file path (directory and file stems). -pub fn path_terms(path: &str) -> Vec { - let mut out = Vec::new(); - for seg in path.split('/') { - let stem = seg.split('.').next().unwrap_or(seg); - for t in tokenize(stem) { - if !out.contains(&t) { - out.push(t); - } - } - } - out -} - -/// Term frequencies for a chunk: body tokens, the symbol name boosted, and -/// path terms once each. -pub fn chunk_terms(body: &str, symbol: Option<&str>, path_terms: &[String]) -> (Vec<(String, u16)>, u32) { - let mut tf: HashMap = HashMap::new(); - let mut len = 0u32; - // Very long chunks (minified or generated code) are capped. - let body = if body.len() > 64 * 1024 { &body[..floor_char(body, 64 * 1024)] } else { body }; - for t in tokenize(body) { - let e = tf.entry(t).or_insert(0); - *e = e.saturating_add(1); - len += 1; - } - if let Some(name) = symbol { - for t in tokenize(name) { - let e = tf.entry(t).or_insert(0); - *e = e.saturating_add(3); - len += 3; - } - } - for t in path_terms { - let e = tf.entry(t.clone()).or_insert(0); - *e = e.saturating_add(1); - len += 1; - } - let mut terms: Vec<(String, u16)> = tf.into_iter().collect(); - terms.sort(); - (terms, len) -} - -fn floor_char(s: &str, mut i: usize) -> usize { - while i > 0 && !s.is_char_boundary(i) { - i -= 1; - } - i -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn splits_identifiers() { - let t = tokenize("parseHTTPResponse user_id XMLParser v2"); - for want in [ - "parsehttpresponse", - "parse", - "http", - "response", - "user_id", - "user", - "id", - "xmlparser", - "xml", - "parser", - "v2", - ] { - assert!(t.contains(&want.to_string()), "missing {want} in {t:?}"); - } - } -} diff --git a/crates/lockdocs/Cargo.toml b/crates/lockdocs/Cargo.toml index 5c8cf57..b1b4227 100644 --- a/crates/lockdocs/Cargo.toml +++ b/crates/lockdocs/Cargo.toml @@ -15,4 +15,4 @@ path = "src/main.rs" anyhow = "1" lockdocs-core = { path = "../lockdocs-core" } serde_json = "1" -sylphx-mcp-kit = { git = "https://github.com/SylphxAI/mcp-kit", tag = "v0.1.0" } +sylphx-mcp-kit = "0.3" diff --git a/crates/lockdocs/src/main.rs b/crates/lockdocs/src/main.rs index 57f96a9..07b13e6 100644 --- a/crates/lockdocs/src/main.rs +++ b/crates/lockdocs/src/main.rs @@ -1,6 +1,5 @@ mod mcp; mod setup; -mod star_hint; mod tools; use anyhow::{bail, Result}; @@ -215,7 +214,12 @@ fn run() -> Result<()> { } }; if result.is_ok() && counts_as_run { - star_hint::after_success(); + mcp_kit::star_hint::after_success( + "Enjoying lockdocs? A GitHub star helps others find it: https://github.com/SylphxAI/lockdocs", + "LOCKDOCS_NO_STAR_HINT", + &lockdocs_core::cache::dir(), + false, + ); } result } diff --git a/crates/lockdocs/src/star_hint.rs b/crates/lockdocs/src/star_hint.rs deleted file mode 100644 index 575a2b3..0000000 --- a/crates/lockdocs/src/star_hint.rs +++ /dev/null @@ -1,134 +0,0 @@ -//! A one-time star line on stderr after the fifth successful CLI run. -//! -//! The run counter lives in the lockdocs cache directory (`LOCKDOCS_CACHE`). The line is shown once -//! ever, and never for the MCP server, for a non-TTY stderr, in CI, or when -//! `LOCKDOCS_NO_STAR_HINT` is set. - -use std::io::IsTerminal; -use std::path::{Path, PathBuf}; - -pub const OPT_OUT_ENV: &str = "LOCKDOCS_NO_STAR_HINT"; -const FILE_NAME: &str = "star-hint"; -const SHOWN: &str = "shown"; -const SHOW_AFTER_RUNS: u32 = 5; -const MESSAGE: &str = "Enjoying lockdocs? A GitHub star helps others find it: https://github.com/SylphxAI/lockdocs"; - -/// What the environment says about where the run is happening. -pub struct Context { - pub stderr_is_tty: bool, - pub ci: bool, - pub opted_out: bool, -} - -impl Context { - fn from_process() -> Self { - let set = |key: &str| std::env::var_os(key).is_some_and(|v| !v.is_empty()); - Context { - stderr_is_tty: std::io::stderr().is_terminal(), - ci: set("CI"), - opted_out: set(OPT_OUT_ENV), - } - } - - fn quiet(&self) -> bool { - !self.stderr_is_tty || self.ci || self.opted_out - } -} - -/// Decide from the stored counter text. Returns the text to store next and -/// whether to print the line now. -pub fn step(stored: Option<&str>, context: &Context) -> (Option, bool) { - if context.quiet() { - return (None, false); - } - let stored = stored.map(str::trim).unwrap_or(""); - if stored == SHOWN { - return (None, false); - } - let runs = stored.parse::().unwrap_or(0).saturating_add(1); - if runs >= SHOW_AFTER_RUNS { - (Some(SHOWN.to_string()), true) - } else { - (Some(runs.to_string()), false) - } -} - -fn state_path() -> PathBuf { - lockdocs_core::cache::dir().join(FILE_NAME) -} - -fn apply(path: &Path, context: &Context) -> bool { - let stored = std::fs::read_to_string(path).ok(); - let (next, show) = step(stored.as_deref(), context); - if let Some(next) = next { - if let Some(parent) = path.parent() { - let _ = std::fs::create_dir_all(parent); - } - if std::fs::write(path, next).is_err() { - return false; - } - } - show -} - -/// Call after a CLI run that succeeded. Best effort: any file error is silent. -pub fn after_success() { - let context = Context::from_process(); - if context.quiet() { - return; - } - if apply(&state_path(), &context) { - eprintln!("{MESSAGE}"); - } -} - -#[cfg(test)] -mod tests { - use super::*; - - fn context(stderr_is_tty: bool, ci: bool, opted_out: bool) -> Context { - Context { stderr_is_tty, ci, opted_out } - } - - #[test] - fn counts_four_runs_then_shows_once() { - let interactive = context(true, false, false); - let mut stored: Option = None; - for run in 1..=4 { - let (next, show) = step(stored.as_deref(), &interactive); - assert!(!show, "run {run}"); - stored = next; - } - assert_eq!(stored.as_deref(), Some("4")); - let (next, show) = step(stored.as_deref(), &interactive); - assert!(show); - assert_eq!(next.as_deref(), Some("shown")); - let (next, show) = step(next.as_deref(), &interactive); - assert!(!show); - assert_eq!(next, None); - } - - #[test] - fn stays_silent_and_uncounted_when_gated() { - for quiet in [context(false, false, false), context(true, true, false), context(true, false, true)] { - assert_eq!(step(Some("4"), &quiet), (None, false)); - } - } - - #[test] - fn garbage_counter_restarts() { - let interactive = context(true, false, false); - assert_eq!(step(Some("x"), &interactive), (Some("1".into()), false)); - } - - #[test] - fn persists_and_shows_on_the_fifth_run_only() { - let dir = std::env::temp_dir().join(format!("lockdocs-star-hint-{}", std::process::id())); - let path = dir.join("nested").join(FILE_NAME); - let interactive = context(true, false, false); - let shown: Vec = (0..7).map(|_| apply(&path, &interactive)).collect(); - assert_eq!(shown, [false, false, false, false, true, false, false]); - assert_eq!(std::fs::read_to_string(&path).unwrap(), "shown"); - let _ = std::fs::remove_dir_all(&dir); - } -} diff --git a/docs/capabilities.md b/docs/capabilities.md index a235a49..d1036de 100644 --- a/docs/capabilities.md +++ b/docs/capabilities.md @@ -13,7 +13,7 @@ is in [vision.md](vision.md). | LD-DOCS-SITE | Add an official docs-site repository for the pinned major: the default branch for the latest major, a `vN` branch or the last commit before the next major for older ones (React, Express, Tailwind CSS, Prisma, tokio) | partial | crates/lockdocs-core/src/upstream.rs | LD-UPSTREAM | | LD-EXTRACT | Split Markdown, MDX, reStructuredText and component-based docs pages into sections; extract TypeScript, Python, Rust and Go symbols with signatures and doc comments | supported | crates/lockdocs-core/src/extract.rs, crates/lockdocs-core/src/markdown.rs | LD-LOCATE | | LD-INDEX | Cache one index per package, version and source location | supported | crates/lockdocs-core/src/index.rs, crates/lockdocs-core/src/cache.rs | LD-EXTRACT | -| LD-SEARCH | Rank sections and symbols: BM25 on full text and on headings, a local embedding model, and docs-specific signals | supported | crates/lockdocs-core/src/bm25.rs, crates/lockdocs-core/src/embed.rs, crates/lockdocs-core/src/query.rs | LD-INDEX | +| LD-SEARCH | Rank sections and symbols: BM25 on full text and on headings, a local embedding model, and docs-specific signals | supported | crates/lockdocs-core/src/bm25.rs, crates/lockdocs-core/src/semantic.rs, crates/lockdocs-core/src/query.rs | LD-INDEX | | LD-MCP | MCP server with the `resolve`, `docs` and `api` tools over stdio | supported | crates/lockdocs/src/mcp.rs, crates/lockdocs/src/tools.rs | LD-SEARCH | | LD-CLI | Command line with the same queries, plus `index`, `fetch` and `setup` | supported | crates/lockdocs/src/main.rs, crates/lockdocs/src/setup.rs | LD-SEARCH | | LD-NPM | npm launcher and native binaries for five platforms | supported | packages/lockdocs, packages/npm | LD-CLI | diff --git a/docs/guide/how-it-works.md b/docs/guide/how-it-works.md index 2be4f67..f3f8ee8 100644 --- a/docs/guide/how-it-works.md +++ b/docs/guide/how-it-works.md @@ -8,3 +8,10 @@ 6. **Pack.** The top result also quotes the first paragraph of its page and parent section (with the short list or code block that follows), when they add something: a deprecation notice at the top of the page, or the setup a subsection builds on. Results are added in rank order until the token budget is spent (the last one trimmed at a line boundary), each with a `package@version path:line` citation. `api` is exact rather than ranked: it finds entries whose name matches the last segment of the symbol, scores qualifiers (`Router` in `axum::Router::route`), follows re-exports, and adds overloads, members and other matches. + +Shared mechanics (lexical tokenization, model download/loading, quantization, +cache-root selection and the one-time CLI star hint) come from +`sylphx-mcp-kit` 0.3. lockdocs keeps its model choice, identifier input policy, +ranking and retention. Existing model files and persisted embedding indexes +remain readable in the same `LOCKDOCS_CACHE` directory; no rebuild or migration +is needed.