diff --git a/Cargo.lock b/Cargo.lock index 100f107..616fef6 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -405,12 +405,24 @@ version = "3.20.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" +[[package]] +name = "bytemuck" +version = "1.25.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" + [[package]] name = "byteorder" version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" +[[package]] +name = "byteorder-lite" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f1fe948ff07f4bd06c30984e69f5b4899c516a3ef74f34df92a2df2ab535495" + [[package]] name = "bytes" version = "1.12.1" @@ -922,6 +934,37 @@ version = "0.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7eed2c4702fa172d1ce21078faa7c5203e69f5394d48cc436d25928394a867a2" +[[package]] +name = "defmt" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e2953bfe4f93bbd20cc71198842756f77d161884c99ebbabc41d80231ded88d1" +dependencies = [ + "bitflags 1.3.2", + "defmt-macros", +] + +[[package]] +name = "defmt-macros" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bad9c72e7ca2137e0dc3813245a0d282fd6daad32fd800af018306a9169b5fe8" +dependencies = [ + "defmt-parser", + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "defmt-parser" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" +dependencies = [ + "thiserror 2.0.18", +] + [[package]] name = "der" version = "0.8.1" @@ -1150,6 +1193,15 @@ version = "2.4.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6" +[[package]] +name = "fdeflate" +version = "0.3.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e6853b52649d4ac5c0bd02320cddc5ba956bdb407c4b75a2c6b75bf51500f8c" +dependencies = [ + "simd-adler32", +] + [[package]] name = "file-id" version = "0.2.3" @@ -1175,6 +1227,16 @@ version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +[[package]] +name = "flatbuffers" +version = "24.12.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4f1baf0dbf96932ec9a3038d57900329c015b0bfb7b63d904f3bc27e2b02a096" +dependencies = [ + "bitflags 1.3.2", + "rustc_version", +] + [[package]] name = "flate2" version = "1.1.9" @@ -1831,6 +1893,21 @@ dependencies = [ "icu_properties", ] +[[package]] +name = "image" +version = "0.25.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85ab80394333c02fe689eaf900ab500fbd0c2213da414687ebf995a65d5a6104" +dependencies = [ + "bytemuck", + "byteorder-lite", + "moxcms", + "num-traits", + "png", + "zune-core", + "zune-jpeg", +] + [[package]] name = "indexmap" version = "1.9.3" @@ -1930,6 +2007,59 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "jiff" +version = "0.2.35" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "668b7183bd07af9a4885f5c35b0cc5c83c4607a913c16b7e17291832910d2dcc" +dependencies = [ + "defmt", + "jiff-core", + "jiff-static", + "jiff-tzdb-platform", + "log", + "portable-atomic", + "portable-atomic-util", + "serde_core", + "windows-link", +] + +[[package]] +name = "jiff-core" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7feca88439efe53da3754500c1851dedf3cb36c524dd5cf8225cc0794de95d09" +dependencies = [ + "defmt", +] + +[[package]] +name = "jiff-static" +version = "0.2.35" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a69dcb3a21cfb32ce1cd056169337ca284af0766dd766e7878819b251a49204" +dependencies = [ + "jiff-core", + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "jiff-tzdb" +version = "0.1.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e" + +[[package]] +name = "jiff-tzdb-platform" +version = "0.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "875a5a69ac2bab1a891711cf5eccbec1ce0341ea805560dcd90b7a2e925132e8" +dependencies = [ + "jiff-tzdb", +] + [[package]] name = "jni" version = "0.22.4" @@ -2099,20 +2229,24 @@ dependencies = [ "aes", "bitflags 2.13.0", "cbc", + "chrono", "ecb", "encoding_rs", "flate2", "getrandom 0.4.3", "indexmap 2.14.0", "itoa", + "jiff", "log", "md-5 0.10.6", "nom 8.0.0", "rand 0.10.2", "rangemap", + "rayon", "sha2 0.10.9", "stringprep", "thiserror 2.0.18", + "time", "ttf-parser", "weezl", ] @@ -2296,6 +2430,16 @@ dependencies = [ "syn", ] +[[package]] +name = "moxcms" +version = "0.8.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb85c154ba489f01b25c0d36ae69a87e4a1c73a72631fc6c0eb6dde34a73e44b" +dependencies = [ + "num-traits", + "pxfm", +] + [[package]] name = "multer" version = "3.1.0" @@ -2504,6 +2648,21 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "ocrs" +version = "0.12.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a5379fdd3f11522b5a2ff53017a189463dabf5d0a9c915cb3eb97fabec4ea11c" +dependencies = [ + "anyhow", + "rayon", + "rten", + "rten-imageproc", + "rten-tensor", + "thiserror 2.0.18", + "wasm-bindgen", +] + [[package]] name = "once_cell" version = "1.21.4" @@ -2731,6 +2890,19 @@ version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" +[[package]] +name = "png" +version = "0.18.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "60769b8b31b2a9f263dae2776c37b1b28ae246943cf719eb6946a1db05128a61" +dependencies = [ + "bitflags 2.13.0", + "crc32fast", + "fdeflate", + "flate2", + "miniz_oxide", +] + [[package]] name = "polyval" version = "0.6.2" @@ -2835,6 +3007,12 @@ dependencies = [ "prost", ] +[[package]] +name = "pxfm" +version = "0.1.30" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d55d956fa96f5ec02be2e13af0e20391a5aa83d6a074e3ad368959d0fab299ea" + [[package]] name = "qdrant-client" version = "1.18.0" @@ -3309,6 +3487,113 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "rten" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "43c230fa4ade87c913f61dbd911b7eb0d49460ceff3f1e4fabc837fac191137c" +dependencies = [ + "flatbuffers", + "num_cpus", + "rayon", + "rten-base", + "rten-gemm", + "rten-model-file", + "rten-onnx", + "rten-shape-inference", + "rten-simd", + "rten-tensor", + "rten-vecmath", + "rustc-hash", + "smallvec", + "typeid", + "wasm-bindgen", +] + +[[package]] +name = "rten-base" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2738cf8bb4c27f828ac788d01ccf4e367e8e773cfec6851f81851b5211de6a79" +dependencies = [ + "rayon", +] + +[[package]] +name = "rten-gemm" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a81a0ca209fb5ce21bd17efa0bd287d5881c6cebfbff0b21c4294a1a14a9e" +dependencies = [ + "rayon", + "rten-base", + "rten-simd", + "rten-tensor", +] + +[[package]] +name = "rten-imageproc" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d5f148e7e941fb5727b9046a5fa1b45525543d5105f14b384fd9261df0ee49bc" +dependencies = [ + "rten-tensor", +] + +[[package]] +name = "rten-model-file" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed2f8d270f07ab1bbfff47250c6039f6caa5da59d6da7d74f66aa48559aa6fea" +dependencies = [ + "flatbuffers", + "rten-base", +] + +[[package]] +name = "rten-onnx" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "23086eef75bfb55278cb0b45cf9f5a877d466d914914aafebee4ffca9b24d20c" + +[[package]] +name = "rten-shape-inference" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e8a913c7ca40e2bfbb2a0cd447cce56b33ab19435f56693271a2ef37cf58984" +dependencies = [ + "rten-tensor", + "smallvec", +] + +[[package]] +name = "rten-simd" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b19a0032dfcb70dd20960c1c51a37674b237586cbc1ce586f45b46605d108e82" + +[[package]] +name = "rten-tensor" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "05dc744a270aa32d154f1a3df8e48740ccc1be9dfbcf23295ada66d83aa98de6" +dependencies = [ + "rayon", + "rten-base", + "smallvec", + "typeid", +] + +[[package]] +name = "rten-vecmath" +version = "0.24.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9574ddebf5671bc08ceb76e2e1638fadc57fdeff318634eab2c29e9a803cff64" +dependencies = [ + "rten-base", + "rten-simd", +] + [[package]] name = "rtoolbox" version = "0.0.5" @@ -4505,6 +4790,12 @@ version = "0.12.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8e28f89b80c87b8fb0cf04ab448d5dd0dd0ade2f8891bae878de66a75a28600e" +[[package]] +name = "typeid" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bc7d623258602320d5c55d1bc22793b57daff0ec7efc270ea7d55ce1d5f5471c" + [[package]] name = "typenum" version = "1.20.1" @@ -4795,17 +5086,22 @@ dependencies = [ "cfb", "chrono", "clap", + "flate2", "futures-util", "hmac 0.12.1", + "image", + "lopdf", "moka", "notify", "notify-debouncer-full", "object_store", + "ocrs", "pdf-extract", "pgvector", "quick-xml 0.41.0", "rand_core 0.6.4", "reqwest 0.12.28", + "rten", "serde", "serde_json", "serde_yaml_ng", @@ -4815,6 +5111,7 @@ dependencies = [ "tokio", "tracing", "tracing-subscriber", + "ureq", "uuid", "verity-core", "verity-encoder", @@ -5535,3 +5832,18 @@ dependencies = [ "log", "simd-adler32", ] + +[[package]] +name = "zune-core" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cb8a0807f7c01457d0379ba880ba6322660448ddebc890ce29bb64da71fb40f9" + +[[package]] +name = "zune-jpeg" +version = "0.5.15" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "27bc9d5b815bc103f142aa054f561d9187d191692ec7c2d1e2b4737f8dbd7296" +dependencies = [ + "zune-core", +] diff --git a/Cargo.toml b/Cargo.toml index 6500a48..686dab4 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -71,6 +71,28 @@ cfb = "0.10" # lopdf — the part we refuse to reimplement. Known to panic on hostile files; # every call is wrapped in catch_unwind (see extract.rs). pdf-extract = "0.12" +# OCR tier (ocr.rs): ocrs + rten — pure-Rust OCR. No tesseract or any system +# dependency (that would break the clean-VM stranger gate), no cloud OCR. +# Models (~12 MB total) download once from the ocrs project's canonical bucket +# into ~/.cache/ocrs — the same fetch-once-then-cache lane as the MiniLM +# encoder (which uses the hf-hub default cache). rten must track the exact +# version the ocrs release depends on, or Model types won't line up. +ocrs = "0.12" +rten = "0.24" +# image: PNG/JPEG decode for standalone image OCR and embedded PDF images, +# plus JPEG *encoding* in test fixtures. Decode-only features, pure Rust. +image = { version = "0.25", default-features = false, features = ["png", "jpeg"] } +# lopdf: pinned to the same version pdf-extract already compiles, so walking a +# scanned PDF's page/image tree reuses the one PDF parser in the tree. +lopdf = "0.42" +# flate2: capped FlateDecode inflation for embedded PDF images (ocr.rs). We +# deliberately do NOT use lopdf's decompressed_content() there — it inflates +# without any output bound, so a 100 KB stream declaring a 10x10 image can +# materialize gigabytes (decompression bomb). Already in the tree via zip. +flate2 = "1" +# ureq: blocking OCR-model download (extraction is synchronous code); already +# in the tree transitively via hf-hub. +ureq = "3" reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls", "multipart"] } schemars = { version = "1", features = ["chrono04"] } # Local-folder watching (folder_watch.rs): cross-platform FS events @@ -78,3 +100,24 @@ schemars = { version = "1", features = ["chrono04"] } # write bursts and mid-write partials fire once, after the file is quiescent. notify = "6" notify-debouncer-full = "0.3" + +# OCR inference is numeric hot-loop work; unoptimized rten is ~50x slower, +# which makes dev-profile scanned-PDF ingestion (and the VERITY_OCR_E2E test) +# crawl. Optimizing JUST the rten family in dev keeps `cargo build` fast for +# our own code while making local OCR usable. Release builds are unaffected. +[profile.dev.package.rten] +opt-level = 3 +[profile.dev.package.rten-base] +opt-level = 3 +[profile.dev.package.rten-gemm] +opt-level = 3 +[profile.dev.package.rten-imageproc] +opt-level = 3 +[profile.dev.package.rten-simd] +opt-level = 3 +[profile.dev.package.rten-tensor] +opt-level = 3 +[profile.dev.package.rten-vecmath] +opt-level = 3 +[profile.dev.package.ocrs] +opt-level = 3 diff --git a/HONESTY.md b/HONESTY.md index f5d3bf4..f160698 100644 --- a/HONESTY.md +++ b/HONESTY.md @@ -98,6 +98,19 @@ source that has **never** heartbeated fails closed while the gate is on (never-s indistinguishable from stalled). With the gate off — the default — assume a stalled connector serves ACLs as stale as the stall is long, and monitor the heartbeats. +## OCR is local and printed-text-grade, not a document-AI service + +Scanned PDFs and PNG/JPEG images are extracted by a fully local, pure-Rust OCR engine +([ocrs](https://github.com/robertknight/ocrs) on rten; ~12 MB of models fetched once into +`~/.cache/ocrs`, no cloud calls, no system dependencies). It reads printed type well and is +**best-effort beyond that**: expect it to miss handwriting, low-resolution scans, stylized +layouts, and non-Latin scripts. Every extraction receipt discloses the method — `pdf-ocr` +(with `pages_ocred`, capped at 50 pages) or `image-ocr` — so OCR-derived text is never +passed off as a document's own text layer, and a file where OCR finds nothing lands +metadata-only with a typed reason rather than silently indexing empty. Encrypted PDFs are +still declined outright. Unsupported embedded encodings (CCITT/JBIG2/JPEG 2000) are +skipped, not guessed at. + ## Derived-visibility on agent writes: lineage is client-DECLARED, not inferred `remember` (`POST /v1/episodes`) now enforces the SPEC §2 intersection diff --git a/crates/verity-mcp/src/main.rs b/crates/verity-mcp/src/main.rs index 53fe1d8..1f688ea 100644 --- a/crates/verity-mcp/src/main.rs +++ b/crates/verity-mcp/src/main.rs @@ -246,8 +246,9 @@ struct IngestTextParams { struct IngestFileParams { /// Scope handle from memory_open_scope. scope_handle: String, - /// Path to a LOCAL UTF-8 text file. Allowed extensions: .txt, .md, - /// .json, .csv, .html. Maximum size: 512 KB. + /// Path to a LOCAL file. Allowed extensions: .txt, .md, .json, .csv, + /// .html (UTF-8 text) and .png, .jpg, .jpeg (server-side local OCR). + /// Maximum size: 512 KB. path: String, /// Entity tags, e.g. ["account:acme-corp"]. Must be inside the scope's /// entity_scope; omit to inherit the whole scope. @@ -375,22 +376,20 @@ impl VerityMcp { } /// Multipart POST /v1/files: fields `scope_handle`, `entities` - /// (comma-separated, only when tags were given), and `file`. + /// (comma-separated, only when tags were given), and `file`. The part is + /// caller-built: text content for the UTF-8 lane, raw bytes + image mime + /// for the OCR lane (the server extracts by magic either way). async fn post_file( &self, scope_handle: String, entities: Option>, - file_name: String, - content: String, + part: reqwest::multipart::Part, ) -> Result { let mut form = reqwest::multipart::Form::new().text("scope_handle", scope_handle); if let Some(entities) = entities.filter(|e| !e.is_empty()) { form = form.text("entities", entities.join(",")); } - form = form.part( - "file", - reqwest::multipart::Part::text(content).file_name(file_name), - ); + form = form.part("file", part); self.proxy(self.http.post(self.endpoint("/v1/files")).multipart(form)) .await } @@ -404,8 +403,12 @@ fn tool_error(msg: impl Into) -> Result { // ---------- local helpers for the ingest tools ---------- -/// Extensions memory_ingest_file accepts (UTF-8 text-like content only). +/// Extensions memory_ingest_file accepts as UTF-8 text. const INGEST_FILE_EXTENSIONS: [&str; 5] = ["txt", "md", "json", "csv", "html"]; +/// Extensions memory_ingest_file accepts as raw image bytes: the server's +/// local OCR tier (extract.rs + ocr.rs) extracts printed text best-effort, +/// disclosed as method "image-ocr" on the receipt. +const INGEST_IMAGE_EXTENSIONS: [&str; 3] = ["png", "jpg", "jpeg"]; /// memory_ingest_file size cap. const MAX_FILE_BYTES: u64 = 512 * 1024; /// memory_ingest_url download cap. @@ -849,7 +852,7 @@ impl VerityMcp { #[tool( name = "memory_ingest_file", - description = "Read a LOCAL text file and ingest its contents into shared memory, so it becomes searchable via memory_recall. Use when the knowledge lives in a file on this machine rather than in-context. Accepts UTF-8 .txt/.md/.json/.csv/.html up to 512 KB; anything else is rejected with an error." + description = "Read a LOCAL file and ingest its contents into shared memory, so it becomes searchable via memory_recall. Use when the knowledge lives in a file on this machine rather than in-context. Accepts UTF-8 text (.txt/.md/.json/.csv/.html) and images (.png/.jpg/.jpeg — printed text is extracted by the server's local OCR, best-effort, disclosed as method image-ocr) up to 512 KB; anything else is rejected with an error." )] async fn memory_ingest_file( &self, @@ -861,9 +864,10 @@ impl VerityMcp { .and_then(|e| e.to_str()) .map(str::to_ascii_lowercase) .unwrap_or_default(); - if !INGEST_FILE_EXTENSIONS.contains(&ext.as_str()) { + let is_image = INGEST_IMAGE_EXTENSIONS.contains(&ext.as_str()); + if !is_image && !INGEST_FILE_EXTENSIONS.contains(&ext.as_str()) { return tool_error(format!( - "unsupported file type {:?}: memory_ingest_file accepts only UTF-8 text files with extension .txt, .md, .json, .csv, or .html", + "unsupported file type {:?}: memory_ingest_file accepts UTF-8 text files (.txt, .md, .json, .csv, .html) and images (.png, .jpg, .jpeg)", p.path )); } @@ -885,6 +889,28 @@ impl VerityMcp { Ok(bytes) => bytes, Err(e) => return tool_error(format!("cannot read file {:?}: {e}", p.path)), }; + if is_image { + // Raw bytes to the server's OCR lane; the server sniffs magic and + // returns a typed, disclosed failure if OCR finds nothing. + let file_name = path + .file_name() + .and_then(|n| n.to_str()) + .unwrap_or("file.png") + .to_owned(); + let mime = if ext == "png" { + "image/png" + } else { + "image/jpeg" + }; + let part = match reqwest::multipart::Part::bytes(bytes) + .file_name(file_name) + .mime_str(mime) + { + Ok(part) => part, + Err(e) => return tool_error(format!("building upload part: {e}")), + }; + return self.post_file(p.scope_handle, p.entities, part).await; + } let content = match String::from_utf8(bytes) { Ok(content) => content, Err(_) => { @@ -899,8 +925,8 @@ impl VerityMcp { .and_then(|n| n.to_str()) .unwrap_or("file.txt") .to_owned(); - self.post_file(p.scope_handle, p.entities, file_name, content) - .await + let part = reqwest::multipart::Part::text(content).file_name(file_name); + self.post_file(p.scope_handle, p.entities, part).await } #[tool( @@ -980,8 +1006,8 @@ impl VerityMcp { return tool_error(format!("no textual content extracted from {url}")); } let file_name = file_name_from_url(&url); - self.post_file(p.scope_handle, p.entities, file_name, content) - .await + let part = reqwest::multipart::Part::text(content).file_name(file_name); + self.post_file(p.scope_handle, p.entities, part).await } #[tool( diff --git a/crates/verity-server/Cargo.toml b/crates/verity-server/Cargo.toml index 8669640..e8435b5 100644 --- a/crates/verity-server/Cargo.toml +++ b/crates/verity-server/Cargo.toml @@ -42,16 +42,37 @@ moka = { workspace = true } # Media object-store seam (task 47): S3/MinIO blob tier behind object_store. object_store = { workspace = true } bytes = "1" -# Tier-1 file extraction (extract.rs): PDF/PPTX/XLS(X) → text, deterministic, -# no OCR. Dep choices justified at the workspace root Cargo.toml. +# Tier-1 file extraction (extract.rs): PDF/PPTX/XLS(X)/DOC(X) text layers, +# deterministic. Dep choices justified at the workspace root Cargo.toml. calamine = { workspace = true } zip = { workspace = true } cfb = { workspace = true } quick-xml = { workspace = true } pdf-extract = { workspace = true } +# OCR tier (ocr.rs): image decode + scanned-PDF page walking are always built +# (typed failures either way); the ocrs/rten engine itself sits behind the +# default-ON `ocr` feature so `--no-default-features` can shave its compile +# cost — extraction then fails typed ("built without the 'ocr' feature"). +image = { workspace = true } +lopdf = { workspace = true } +# Bomb-guarded FlateDecode for embedded PDF images: lopdf's own +# decompressed_content() has no output cap, so ocr.rs inflates manually with +# flate2 + Read::take. Non-optional — the decode path compiles either way. +flate2 = { workspace = true } +ocrs = { workspace = true, optional = true } +rten = { workspace = true, optional = true } +ureq = { workspace = true, optional = true } # SSE subscriptions (task 21): poll-loop event streams. async-stream = "0.3" futures-util = { version = "0.3", default-features = false, features = ["alloc"] } # Local-folder watching (folder_watch.rs): server-side FS watcher + debounce. notify = { workspace = true } notify-debouncer-full = { workspace = true } + +[features] +default = ["ocr"] +# Local OCR engine (ocrs + rten). ON by default — sovereignty-first, still +# pure Rust, still zero system deps. Building with --no-default-features +# drops the engine (and its compile time); scanned-PDF/image extraction then +# returns the typed "built without the 'ocr' feature" failure instead. +ocr = ["dep:ocrs", "dep:rten", "dep:ureq"] diff --git a/crates/verity-server/src/extract.rs b/crates/verity-server/src/extract.rs index 18c0446..a8fbbf7 100644 --- a/crates/verity-server/src/extract.rs +++ b/crates/verity-server/src/extract.rs @@ -1,19 +1,26 @@ -//! Tier-1 binary file extraction: PDF, PPTX, XLS(X) → plain text. +//! Tier-1 binary file extraction: PDF, PPTX, XLS(X), DOC(X) → plain text, +//! plus the local OCR tier for scanned PDFs and PNG/JPEG images (ocr.rs). //! //! The honesty rules here are load-bearing (founder directive, Tier 1): //! -//! * **Rust-native and deterministic.** No LLM, no OCR, no external process. -//! The same bytes always yield the same text. +//! * **Rust-native, local, no external process.** No LLM, no cloud OCR, no +//! system dependency. Text-layer extraction is fully deterministic; the OCR +//! paths ("pdf-ocr" / "image-ocr") are printed-text-grade and BEST-EFFORT — +//! the receipt always discloses the method so a consumer can weigh +//! OCR-derived text accordingly (same bytes + same cached models still +//! yield the same text; a model update may change it). //! * **Never silently empty.** Every call returns either extracted text with //! its method + truncation flag, a *typed* failure reason (encrypted PDF, -//! scanned/image PDF, parse failure, unrecognized format), or an explicit -//! `NotHandled` so the caller can run its existing text-like/store-only -//! logic. A caller can always disclose exactly what happened. -//! * **A hostile file never kills the server.** The PDF path is wrapped in +//! scanned PDF where OCR found nothing, OCR engine unavailable, parse +//! failure, unrecognized format), or an explicit `NotHandled` so the caller +//! can run its existing text-like/store-only logic. A caller can always +//! disclose exactly what happened. +//! * **A hostile file never kills the server.** The PDF paths are wrapped in //! `catch_unwind` because pdf crates are known to panic on malformed input. //! * **Honest limits.** Extraction is capped at [`MAX_EXTRACT_CHARS`] with a -//! disclosed `truncated` flag; scanned PDFs are declined with an explicit -//! "OCR is a later tier" reason, not returned as empty text. +//! disclosed `truncated` flag; scanned-PDF OCR is capped at +//! `ocr::MAX_OCR_PAGES` pages with the page count disclosed via +//! `pages_ocred`; encrypted PDFs are still declined outright. //! //! Dependency choices (also noted in the workspace Cargo.toml): //! @@ -33,6 +40,8 @@ use std::io::{Cursor, Read}; use calamine::Reader as _; +use crate::ocr::{self, OcrBackend}; + /// Hard cap on extracted text, in chars (~200 KB). Disclosed via /// `Extraction::truncated` — a capped extraction is never passed off as the /// whole document. @@ -43,9 +52,34 @@ pub(crate) const MAX_EXTRACT_CHARS: usize = 200_000; #[derive(Debug)] pub(crate) struct Extraction { pub(crate) text: String, - /// "calamine" | "pptx-xml" | "pdf-text" — recorded into provenance. + /// "calamine" | "pptx-xml" | "docx-xml" | "doc-piecetable" | "pdf-text" | + /// "pdf-ocr" | "image-ocr" — recorded into provenance. The two `-ocr` + /// methods mark best-effort recognized text, not a deterministic text + /// layer. pub(crate) method: &'static str, pub(crate) truncated: bool, + /// For "pdf-ocr" only: the honest page accounting — pages OCRed (engine + /// consulted; capped at [`ocr::MAX_OCR_PAGES`]), total pages in the + /// document, and pages skipped because their images use encodings we + /// don't implement. A partial pass is visible on the receipt. + pub(crate) ocr_pages: Option, +} + +/// The disclosed extraction receipt, embedded verbatim in episode payloads +/// and HTTP responses. ONE builder so no call site forgets the OCR page +/// accounting. +pub(crate) fn receipt_json( + method: &str, + truncated: bool, + ocr_pages: Option, +) -> serde_json::Value { + let mut v = serde_json::json!({ "method": method, "truncated": truncated }); + if let Some(pages) = ocr_pages { + v["pages_ocred"] = pages.ocred.into(); + v["pages_total"] = pages.total.into(); + v["pages_skipped_unsupported"] = pages.skipped_unsupported.into(); + } + v } /// Typed failure reasons. `reason()` strings are part of the disclosed API @@ -53,14 +87,31 @@ pub(crate) struct Extraction { /// them deliberately. #[derive(Debug, PartialEq, Eq)] pub(crate) enum ExtractFailure { + /// Encrypted PDFs are declined outright — no decryption attempts, no OCR. EncryptedPdf, - /// Parsed fine but produced (approximately) no text: an image-only scan. + /// No text layer AND the OCR pass recognized nothing (or found no + /// decodable raster images at all). ScannedPdf, + /// No text layer, and every image-bearing page uses an encoding we don't + /// implement (CCITT/JBIG2/JPX, exotic colorspaces): OCR never got to + /// ATTEMPT recognition. Distinct from [`Self::ScannedPdf`], which means + /// OCR ran (or had nothing at all to run on) and found none — conflating + /// the two would pass off "we can't read this format" as "there was + /// nothing to read". + UnsupportedPdfImages, PdfParse(String), SheetParse(String), PptxParse(String), DocxParse(String), DocParse(String), + /// A PNG/JPEG whose bytes would not decode. + ImageParse(String), + /// An image parsed fine but OCR recognized no text in it. + ImageNoText, + /// The local OCR engine could not run (model download/init/inference + /// failed, or the server was built without the `ocr` feature). Typed and + /// disclosed — never a panic, never silent empty. + OcrUnavailable(String), /// The filename claimed one of our formats but the bytes don't match /// (magic wins), or the connector sent bytes we have no extractor for. UnrecognizedFormat, @@ -73,12 +124,18 @@ impl ExtractFailure { pub(crate) fn reason(&self) -> String { match self { Self::EncryptedPdf => "encrypted PDF".into(), - Self::ScannedPdf => "scanned/image PDF — no text layer (OCR is a later tier)".into(), + Self::ScannedPdf => "scanned/image PDF — no text layer and OCR found none".into(), + Self::UnsupportedPdfImages => { + "scanned/image PDF — unsupported image encodings; OCR could not attempt".into() + } Self::PdfParse(e) => format!("PDF parse failure: {e}"), Self::SheetParse(e) => format!("spreadsheet parse failure: {e}"), Self::PptxParse(e) => format!("PPTX parse failure: {e}"), Self::DocxParse(e) => format!("DOCX parse failure: {e}"), Self::DocParse(e) => format!("legacy .doc parse failure: {e}"), + Self::ImageParse(e) => format!("image decode failure: {e}"), + Self::ImageNoText => "image parsed but OCR found no text".into(), + Self::OcrUnavailable(e) => format!("OCR unavailable: {e}"), Self::UnrecognizedFormat => "unrecognized format".into(), Self::NoText => "file parsed but contains no extractable text".into(), } @@ -110,6 +167,8 @@ enum Claim { Xls, Docx, Doc, + Png, + Jpeg, } fn claim_from_name(filename: Option<&str>) -> Option { @@ -126,6 +185,10 @@ fn claim_from_name(filename: Option<&str>) -> Option { Some(Claim::Docx) } else if lower.ends_with(".doc") { Some(Claim::Doc) + } else if lower.ends_with(".png") { + Some(Claim::Png) + } else if lower.ends_with(".jpg") || lower.ends_with(".jpeg") { + Some(Claim::Jpeg) } else { None } @@ -133,6 +196,9 @@ fn claim_from_name(filename: Option<&str>) -> Option { const ZIP_MAGIC: &[u8] = b"PK\x03\x04"; const OLE2_MAGIC: &[u8] = &[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1]; +const PNG_MAGIC: &[u8] = b"\x89PNG\r\n\x1a\n"; +/// JPEG/JFIF/EXIF all open with the SOI marker followed by another marker. +const JPEG_MAGIC: &[u8] = &[0xFF, 0xD8, 0xFF]; /// The PDF spec allows up to 1024 bytes of junk before `%PDF-`. fn looks_like_pdf(bytes: &[u8]) -> bool { @@ -154,9 +220,24 @@ pub(crate) fn extract(bytes: &[u8], filename: Option<&str>) -> ExtractOutcome { /// Cap-parameterized worker so tests can exercise truncation without /// megabyte fixtures. Production callers use [`extract`]. fn extract_with_cap(bytes: &[u8], filename: Option<&str>, cap: usize) -> ExtractOutcome { + extract_with_ocr(bytes, filename, cap, ocr::default_backend()) +} + +/// Backend-parameterized worker: tests inject fake OCR backends here so the +/// OCR plumbing (routing, page walking, caps, failure taxonomy) is provable +/// hermetically — no model downloads in unit tests. +fn extract_with_ocr( + bytes: &[u8], + filename: Option<&str>, + cap: usize, + ocr: &dyn OcrBackend, +) -> ExtractOutcome { let claim = claim_from_name(filename); if looks_like_pdf(bytes) { - return finish(extract_pdf(bytes, cap)); + return finish(extract_pdf(bytes, cap, ocr)); + } + if bytes.starts_with(PNG_MAGIC) || bytes.starts_with(JPEG_MAGIC) { + return finish(extract_image(bytes, cap, ocr)); } if bytes.starts_with(ZIP_MAGIC) { // Zip container: xlsx and pptx are both zips — tell them apart by the @@ -168,10 +249,7 @@ fn extract_with_cap(bytes: &[u8], filename: Option<&str>, cap: usize) -> Extract // A zip that isn't an office package: ours only if the name // claimed so (then the claim is wrong — typed, magic wins). Ok(_) => match claim { - Some(Claim::Xlsx) | Some(Claim::Pptx) | Some(Claim::Docx) | Some(Claim::Xls) - | Some(Claim::Doc) | Some(Claim::Pdf) => { - ExtractOutcome::Failed(ExtractFailure::UnrecognizedFormat) - } + Some(_) => ExtractOutcome::Failed(ExtractFailure::UnrecognizedFormat), None => ExtractOutcome::NotHandled, }, Err(e) => match claim { @@ -180,7 +258,7 @@ fn extract_with_cap(bytes: &[u8], filename: Option<&str>, cap: usize) -> Extract Some(Claim::Xlsx) | Some(Claim::Xls) => { ExtractOutcome::Failed(ExtractFailure::SheetParse(e)) } - Some(Claim::Pdf) | Some(Claim::Doc) => { + Some(Claim::Pdf) | Some(Claim::Doc) | Some(Claim::Png) | Some(Claim::Jpeg) => { ExtractOutcome::Failed(ExtractFailure::UnrecognizedFormat) } None => ExtractOutcome::NotHandled, @@ -249,7 +327,10 @@ fn zip_office_kind(bytes: &[u8]) -> Result, String> { // Char-budgeted output assembly (shared truncation semantics) // --------------------------------------------------------------------------- -struct Budget { +/// pub(crate): ocr.rs pushes OCRed page blocks through the same budget so +/// pdf-ocr shares the exact truncation semantics (and stops OCRing early once +/// the cap is hit — recognition is the expensive part). +pub(crate) struct Budget { out: String, remaining: usize, truncated: bool, @@ -267,7 +348,7 @@ impl Budget { /// Append `s`; if the budget runs out mid-string, cut at a char boundary /// and mark truncated. Returns false once the budget is exhausted so /// producers can stop early instead of materializing unbounded text. - fn push(&mut self, s: &str) -> bool { + pub(crate) fn push(&mut self, s: &str) -> bool { if self.truncated { return false; } @@ -294,6 +375,7 @@ impl Budget { text: self.out, method, truncated: self.truncated, + ocr_pages: None, } } } @@ -350,6 +432,7 @@ fn extract_sheet(bytes: &[u8], cap: usize) -> Result text: String::new(), // folded to ExtractFailure::NoText by finish() method: "calamine", truncated: false, + ocr_pages: None, }); } Ok(budget.into_extraction("calamine")) @@ -427,6 +510,7 @@ fn extract_pptx(bytes: &[u8], cap: usize) -> Result text: String::new(), // folded to ExtractFailure::NoText by finish() method: "pptx-xml", truncated: false, + ocr_pages: None, }); } Ok(budget.into_extraction("pptx-xml")) @@ -496,6 +580,7 @@ fn extract_docx(bytes: &[u8], cap: usize) -> Result text: budget.out, method: "docx-xml", truncated: budget.truncated, + ocr_pages: None, }) } @@ -653,6 +738,7 @@ fn extract_doc(bytes: &[u8], cap: usize) -> Result { text: budget.out.trim().to_string(), method: "doc-piecetable", truncated: budget.truncated, + ocr_pages: None, }) } @@ -698,10 +784,15 @@ fn notes_target(rels_xml: &str) -> Option { } // --------------------------------------------------------------------------- -// PDF via pdf-extract (text layer only — no OCR in Tier 1) +// PDF via pdf-extract (text layer), falling back to local OCR (ocr.rs) when +// the document has none // --------------------------------------------------------------------------- -fn extract_pdf(bytes: &[u8], cap: usize) -> Result { +fn extract_pdf( + bytes: &[u8], + cap: usize, + ocr: &dyn OcrBackend, +) -> Result { // Encryption check FIRST, on the raw bytes: the `/Encrypt` key legitimately // appears only in the trailer dictionary. A false positive would require // the literal token in an *uncompressed* content stream — vanishingly rare @@ -726,15 +817,92 @@ fn extract_pdf(bytes: &[u8], cap: usize) -> Result { } }; if text.trim().is_empty() { - // Parsed fine, ~no text: an image-only/scanned PDF. Tier 1 has no OCR - // — disclose that instead of indexing nothing silently. - return Err(ExtractFailure::ScannedPdf); + // Parsed fine, ~no text layer: a scanned/image PDF. Run the local OCR + // pass (ocr.rs) over its embedded page images — best-effort, + // disclosed as "pdf-ocr" with pages_ocred, and still a typed failure + // when OCR finds nothing or cannot run. Fenced like the text pass: + // lopdf must never unwind into the handler either. + return ocr_scanned_pdf(bytes, cap, ocr); } let mut budget = Budget::new(cap); budget.push(text.trim()); Ok(budget.into_extraction("pdf-text")) } +/// The scanned-PDF OCR pass. The engine is only consulted when a page image +/// actually decodes, so a blank or vector-only PDF fails fast as ScannedPdf +/// without touching (or downloading) any model. +fn ocr_scanned_pdf( + bytes: &[u8], + cap: usize, + ocr: &dyn OcrBackend, +) -> Result { + let outcome = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let mut budget = Budget::new(cap); + let pages = ocr::ocr_pdf_pages(bytes, &mut budget, ocr::MAX_OCR_PAGES, ocr)?; + Ok((budget, pages)) + })); + let (budget, pages) = match outcome { + Ok(Ok(ok)) => ok, + Ok(Err(ocr::PdfOcrError::Parse(e))) => { + return Err(ExtractFailure::PdfParse(format!("OCR pass: {e}"))) + } + Ok(Err(ocr::PdfOcrError::Engine(e))) => return Err(ExtractFailure::OcrUnavailable(e)), + Err(_) => { + return Err(ExtractFailure::PdfParse( + "OCR pass: parser panicked on malformed input (contained)".into(), + )) + } + }; + if budget.out.trim().is_empty() { + // Honesty split: if OCR never got to attempt a single page because + // every image-bearing page uses an encoding we don't implement, say + // THAT — "OCR found none" would be a false claim of having looked. + if pages.ocred == 0 && pages.skipped_unsupported > 0 { + return Err(ExtractFailure::UnsupportedPdfImages); + } + return Err(ExtractFailure::ScannedPdf); + } + let mut ex = budget.into_extraction("pdf-ocr"); + ex.ocr_pages = Some(pages); + Ok(ex) +} + +// --------------------------------------------------------------------------- +// Standalone PNG/JPEG via local OCR (ocr.rs) +// --------------------------------------------------------------------------- + +fn extract_image( + bytes: &[u8], + cap: usize, + ocr: &dyn OcrBackend, +) -> Result { + // Fenced like the PDF lanes: a decoder or engine panic on a hostile image + // must surface as a typed failure, never unwind into the handler (where + // it would become a JoinError 500 off the blocking pool). + let outcome = std::panic::catch_unwind(std::panic::AssertUnwindSafe( + || -> Result { + let img = ocr::decode_rgb(bytes).map_err(ExtractFailure::ImageParse)?; + let text = ocr + .recognize_rgb(img.width(), img.height(), img.as_raw()) + .map_err(ExtractFailure::OcrUnavailable)?; + let text = text.trim(); + if text.is_empty() { + return Err(ExtractFailure::ImageNoText); + } + let mut budget = Budget::new(cap); + budget.push(text); + Ok(budget.into_extraction("image-ocr")) + }, + )); + match outcome { + Ok(res) => res, + Err(_) => Err(ExtractFailure::ImageParse( + "image decode/OCR panicked on malformed input (contained)".into(), + )), + } +} + // --------------------------------------------------------------------------- // Programmatic fixtures (tests only): tiny valid files, no binaries in-repo. // --------------------------------------------------------------------------- @@ -959,11 +1127,199 @@ pub(crate) mod fixtures { pdf.into_bytes() } - /// A valid PDF with one empty page — parses fine, zero text ops. The - /// stand-in for a scanned/image-only document. + /// A valid PDF with one empty page — parses fine, zero text ops, zero + /// images. The stand-in for a blank/vector-only document: the OCR pass + /// finds nothing to even attempt. pub(crate) fn image_only_pdf() -> Vec { text_pdf(&[]) } + + /// Flat-color JPEG bytes (a valid, decodable page image; the injected OCR + /// fakes don't look at the pixels). + pub(crate) fn jpeg_bytes(w: u32, h: u32) -> Vec { + let img = image::RgbImage::from_pixel(w, h, image::Rgb([180, 180, 180])); + let mut out = Vec::new(); + image::codecs::jpeg::JpegEncoder::new(&mut out) + .encode(img.as_raw(), w, h, image::ExtendedColorType::Rgb8) + .expect("fixture jpeg encodes"); + out + } + + /// Flat-color PNG bytes. + pub(crate) fn png_bytes(w: u32, h: u32) -> Vec { + let img = image::RgbImage::from_pixel(w, h, image::Rgb([64, 64, 64])); + let mut out = std::io::Cursor::new(Vec::new()); + img.write_to(&mut out, image::ImageFormat::Png) + .expect("fixture png encodes"); + out.into_inner() + } + + /// A minimal scanned-style PDF: one page per JPEG, each page's content + /// stream drawing its image XObject (DCTDecode — the stream body IS the + /// JPEG). Exactly the shape scanner exports take: valid structure, zero + /// text operators. Authored programmatically with correct xref offsets. + pub(crate) fn scanned_pdf_with_jpegs(jpegs: &[&[u8]]) -> Vec { + let n_pages = jpegs.len(); + let kids: Vec = (0..n_pages).map(|i| format!("{} 0 R", 3 + i * 3)).collect(); + let mut objects: Vec> = vec![ + b"<< /Type /Catalog /Pages 2 0 R >>".to_vec(), + format!( + "<< /Type /Pages /Kids [{}] /Count {n_pages} >>", + kids.join(" ") + ) + .into_bytes(), + ]; + for (i, jpeg) in jpegs.iter().enumerate() { + use image::GenericImageView as _; + // Page object ids run 3, 6, 9, … (matching `kids` above), each + // followed by its contents and image objects. + let (contents_obj, image_obj) = (4 + i * 3, 5 + i * 3); + let img = image::load_from_memory(jpeg).expect("fixture jpeg decodes"); + let (w, h) = img.dimensions(); + let content = format!("q {w} 0 0 {h} 0 0 cm /Im0 Do Q\n"); + objects.push( + format!( + "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] \ + /Contents {contents_obj} 0 R \ + /Resources << /XObject << /Im0 {image_obj} 0 R >> >> >>" + ) + .into_bytes(), + ); + objects.push( + format!( + "<< /Length {} >>\nstream\n{content}endstream", + content.len() + ) + .into_bytes(), + ); + let mut image_object = format!( + "<< /Type /XObject /Subtype /Image /Width {w} /Height {h} \ + /ColorSpace /DeviceRGB /BitsPerComponent 8 /Filter /DCTDecode \ + /Length {} >>\nstream\n", + jpeg.len() + ) + .into_bytes(); + image_object.extend_from_slice(jpeg); + image_object.extend_from_slice(b"\nendstream"); + objects.push(image_object); + } + + let mut pdf: Vec = b"%PDF-1.4\n".to_vec(); + let mut offsets = Vec::new(); + for (i, obj) in objects.iter().enumerate() { + offsets.push(pdf.len()); + pdf.extend_from_slice(format!("{} 0 obj\n", i + 1).as_bytes()); + pdf.extend_from_slice(obj); + pdf.extend_from_slice(b"\nendobj\n"); + } + let xref_at = pdf.len(); + pdf.extend_from_slice(format!("xref\n0 {}\n", objects.len() + 1).as_bytes()); + pdf.extend_from_slice(b"0000000000 65535 f \n"); + for off in offsets { + pdf.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes()); + } + pdf.extend_from_slice( + format!( + "trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{xref_at}\n%%EOF\n", + objects.len() + 1 + ) + .as_bytes(), + ); + pdf + } + + /// The hostile-image harness: a one-page PDF embedding a single image + /// XObject with EXACTLY the dict entries given (everything but /Length, + /// which is derived) over an arbitrary stream body. Lets tests declare + /// lying dimensions, decompression bombs, and unsupported filters that no + /// honest encoder would produce. + pub(crate) fn pdf_with_image_xobject(image_dict: &str, stream: &[u8]) -> Vec { + let content = "q 100 0 0 100 0 0 cm /Im0 Do Q\n"; + let mut image_object = format!( + "<< /Type /XObject /Subtype /Image {image_dict} /Length {} >>\nstream\n", + stream.len() + ) + .into_bytes(); + image_object.extend_from_slice(stream); + image_object.extend_from_slice(b"\nendstream"); + let objects: Vec> = vec![ + b"<< /Type /Catalog /Pages 2 0 R >>".to_vec(), + b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>".to_vec(), + b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] \ + /Contents 4 0 R /Resources << /XObject << /Im0 5 0 R >> >> >>" + .to_vec(), + format!( + "<< /Length {} >>\nstream\n{content}endstream", + content.len() + ) + .into_bytes(), + image_object, + ]; + + let mut pdf: Vec = b"%PDF-1.4\n".to_vec(); + let mut offsets = Vec::new(); + for (i, obj) in objects.iter().enumerate() { + offsets.push(pdf.len()); + pdf.extend_from_slice(format!("{} 0 obj\n", i + 1).as_bytes()); + pdf.extend_from_slice(obj); + pdf.extend_from_slice(b"\nendobj\n"); + } + let xref_at = pdf.len(); + pdf.extend_from_slice(format!("xref\n0 {}\n", objects.len() + 1).as_bytes()); + pdf.extend_from_slice(b"0000000000 65535 f \n"); + for off in offsets { + pdf.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes()); + } + pdf.extend_from_slice( + format!( + "trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{xref_at}\n%%EOF\n", + objects.len() + 1 + ) + .as_bytes(), + ); + pdf + } + + /// zlib-compress bytes (FlateDecode test payloads). + pub(crate) fn zlib(bytes: &[u8]) -> Vec { + use std::io::Write as _; + let mut enc = flate2::write::ZlibEncoder::new(Vec::new(), flate2::Compression::default()); + enc.write_all(bytes).expect("zlib write"); + enc.finish().expect("zlib finish") + } + + /// The review's probe, as a fixture: a small FlateDecode stream whose + /// dict declares a tiny image but whose payload inflates to ~100 MB. + pub(crate) fn flate_bomb_pdf() -> Vec { + use std::io::Write as _; + let mut enc = flate2::write::ZlibEncoder::new(Vec::new(), flate2::Compression::default()); + let zeros = [0u8; 65536]; + for _ in 0..1600 { + enc.write_all(&zeros).expect("zlib write"); // 1600 * 64 KiB = 100 MiB + } + let bomb = enc.finish().expect("zlib finish"); + assert!(bomb.len() < 256 * 1024, "bomb must be small on the wire"); + pdf_with_image_xobject( + "/Width 10 /Height 10 /ColorSpace /DeviceRGB /BitsPerComponent 8 \ + /Filter /FlateDecode", + &bomb, + ) + } + + /// Patch a fixture JPEG's SOF0 header to claim absurd dimensions — the + /// bytes still decode as a JPEG *header*, but any pixel decode would try + /// to materialize gigapixels. + pub(crate) fn jpeg_with_lying_dims(w: u32, h: u32, claim_w: u16, claim_h: u16) -> Vec { + let mut jpeg = jpeg_bytes(w, h); + let sof = jpeg + .windows(2) + .position(|m| m == [0xFF, 0xC0]) + .expect("baseline fixture JPEG has an SOF0 marker"); + // SOF0: FF C0 len(2) precision(1) height(2) width(2) … + jpeg[sof + 5..sof + 7].copy_from_slice(&claim_h.to_be_bytes()); + jpeg[sof + 7..sof + 9].copy_from_slice(&claim_w.to_be_bytes()); + jpeg + } } // --------------------------------------------------------------------------- @@ -974,6 +1330,32 @@ pub(crate) mod fixtures { mod tests { use super::*; + /// Injected OCR fakes: the plumbing (routing, page walk, caps, failure + /// taxonomy) is proven hermetically — no model downloads in unit tests. + /// Real-engine coverage is the VERITY_OCR_E2E=1-gated test below. + struct FakeOcr(&'static str); + impl OcrBackend for FakeOcr { + fn recognize_rgb(&self, _w: u32, _h: u32, _rgb: &[u8]) -> Result { + Ok(self.0.to_string()) + } + } + + /// The model-unavailable path (download/init failed). + struct DownOcr; + impl OcrBackend for DownOcr { + fn recognize_rgb(&self, _w: u32, _h: u32, _rgb: &[u8]) -> Result { + Err("models unavailable (test)".into()) + } + } + + /// Proves a code path never consults the engine at all. + struct PanicOcr; + impl OcrBackend for PanicOcr { + fn recognize_rgb(&self, _w: u32, _h: u32, _rgb: &[u8]) -> Result { + panic!("OCR backend must not be consulted on this path") + } + } + fn expect_extracted(outcome: ExtractOutcome) -> Extraction { match outcome { ExtractOutcome::Extracted(ex) => ex, @@ -1147,12 +1529,496 @@ mod tests { } #[test] - fn image_only_pdf_is_declined_as_scanned_no_ocr() { - let f = expect_failed(extract(&fixtures::image_only_pdf(), Some("scan.pdf"))); + fn imageless_scanned_pdf_fails_typed_without_consulting_the_engine() { + // No text layer AND no decodable page images: the honest failure, and + // the engine (PanicOcr) is provably never touched — no model download + // is ever triggered by a blank/vector-only PDF. + let f = expect_failed(extract_with_ocr( + &fixtures::image_only_pdf(), + Some("scan.pdf"), + MAX_EXTRACT_CHARS, + &PanicOcr, + )); assert_eq!(f, ExtractFailure::ScannedPdf); assert_eq!( f.reason(), - "scanned/image PDF — no text layer (OCR is a later tier)" + "scanned/image PDF — no text layer and OCR found none" + ); + } + + // ---------------- OCR tier (injected backends; see ocr.rs) ---------------- + + #[test] + fn scanned_pdf_ocrs_pages_in_order_with_pdf_ocr_receipt() { + let j1 = fixtures::jpeg_bytes(24, 16); + let j2 = fixtures::jpeg_bytes(16, 24); + let bytes = fixtures::scanned_pdf_with_jpegs(&[&j1, &j2]); + let ex = expect_extracted(extract_with_ocr( + &bytes, + Some("scan.pdf"), + MAX_EXTRACT_CHARS, + &FakeOcr("the falcon codeword is zanzibar"), + )); + assert_eq!(ex.method, "pdf-ocr"); + assert_eq!( + ex.ocr_pages, + Some(crate::ocr::OcrPages { + ocred: 2, + total: 2, + skipped_unsupported: 0 + }) + ); + assert!(!ex.truncated); + assert!(ex.text.contains("Page 1:")); + assert!(ex.text.contains("Page 2:")); + assert!(ex.text.contains("the falcon codeword is zanzibar")); + let p1 = ex.text.find("Page 1:").unwrap(); + let p2 = ex.text.find("Page 2:").unwrap(); + assert!(p1 < p2, "pages must join in order"); + } + + #[test] + fn scanned_pdf_ocr_honors_the_page_cap() { + let jpeg = fixtures::jpeg_bytes(8, 8); + let pages: Vec<&[u8]> = std::iter::repeat(jpeg.as_slice()) + .take(crate::ocr::MAX_OCR_PAGES + 2) + .collect(); + let bytes = fixtures::scanned_pdf_with_jpegs(&pages); + let ex = expect_extracted(extract_with_ocr( + &bytes, + Some("big-scan.pdf"), + MAX_EXTRACT_CHARS, + &FakeOcr("line"), + )); + assert_eq!(ex.method, "pdf-ocr"); + let pages = ex.ocr_pages.expect("pdf-ocr carries page accounting"); + assert_eq!(pages.ocred, crate::ocr::MAX_OCR_PAGES as u32); + assert_eq!( + pages.total, + crate::ocr::MAX_OCR_PAGES as u32 + 2, + "total must show the WHOLE document so the cap is visible" + ); + assert!(ex + .text + .contains(&format!("Page {}:", crate::ocr::MAX_OCR_PAGES))); + assert!(!ex + .text + .contains(&format!("Page {}:", crate::ocr::MAX_OCR_PAGES + 1))); + } + + #[test] + fn scanned_pdf_ocr_honors_the_char_cap_with_truncated_flag() { + let jpeg = fixtures::jpeg_bytes(8, 8); + let bytes = fixtures::scanned_pdf_with_jpegs(&[&jpeg, &jpeg, &jpeg]); + let ExtractOutcome::Extracted(ex) = extract_with_ocr( + &bytes, + Some("scan.pdf"), + 24, + &FakeOcr("0123456789abcdefghij"), + ) else { + panic!("expected Extracted"); + }; + assert_eq!(ex.method, "pdf-ocr"); + assert!( + ex.truncated, + "char-cap overflow must set the truncated flag" + ); + assert!(ex.text.chars().count() <= 24); + } + + #[test] + fn scanned_pdf_with_ocr_engine_down_is_a_typed_ocr_unavailable_failure() { + let jpeg = fixtures::jpeg_bytes(8, 8); + let bytes = fixtures::scanned_pdf_with_jpegs(&[&jpeg]); + let f = expect_failed(extract_with_ocr( + &bytes, + Some("scan.pdf"), + MAX_EXTRACT_CHARS, + &DownOcr, + )); + assert_eq!( + f, + ExtractFailure::OcrUnavailable("models unavailable (test)".into()) + ); + assert_eq!(f.reason(), "OCR unavailable: models unavailable (test)"); + } + + #[test] + fn scanned_pdf_whose_ocr_finds_nothing_fails_as_scanned_pdf() { + let jpeg = fixtures::jpeg_bytes(8, 8); + let bytes = fixtures::scanned_pdf_with_jpegs(&[&jpeg]); + let f = expect_failed(extract_with_ocr( + &bytes, + Some("scan.pdf"), + MAX_EXTRACT_CHARS, + &FakeOcr(" "), + )); + assert_eq!(f, ExtractFailure::ScannedPdf); + } + + // ------- hostile embedded images: bombs, lying headers, unsupported ------- + + /// Records exactly what bitmaps the engine was shown (pixel-level proof + /// for the decode plumbing) and returns fixed text. + struct CaptureOcr(std::sync::Mutex)>>); + impl CaptureOcr { + fn new() -> Self { + Self(std::sync::Mutex::new(Vec::new())) + } + } + impl OcrBackend for CaptureOcr { + fn recognize_rgb(&self, w: u32, h: u32, rgb: &[u8]) -> Result { + self.0.lock().unwrap().push((w, h, rgb.to_vec())); + Ok("captured".into()) + } + } + + #[test] + fn extract_pdf_flate_bomb_is_skipped_typed_and_never_reaches_the_engine() { + // The review's probe: ~100 KB on the wire, declares 10x10, inflates + // to 100 MB. The capped inflate must skip it (typed ScannedPdf, since + // nothing else is on the page) without materializing the payload and + // without ever consulting the engine. + let bytes = fixtures::flate_bomb_pdf(); + let f = expect_failed(extract_with_ocr( + &bytes, + Some("bomb.pdf"), + MAX_EXTRACT_CHARS, + &PanicOcr, + )); + assert_eq!(f, ExtractFailure::ScannedPdf); + // And nothing was OCRed / nothing counted as "unsupported encoding": + // the bomb is bad DATA on a supported path, not an unsupported format. + let mut budget = Budget::new(MAX_EXTRACT_CHARS); + let pages = + crate::ocr::ocr_pdf_pages(&bytes, &mut budget, crate::ocr::MAX_OCR_PAGES, &PanicOcr) + .expect("walk parses"); + assert_eq!( + pages, + crate::ocr::OcrPages { + ocred: 0, + total: 1, + skipped_unsupported: 0 + } + ); + } + + #[test] + fn extract_inflate_capped_bounds_output_memory_at_the_cap() { + // Seam-level proof of the memory bound: a stream inflating to 10 MB + // against a 1000-byte cap is refused, and the refusal path can never + // have held more than cap+1 bytes of inflated output. + let payload = fixtures::zlib(&vec![0u8; 10 * 1024 * 1024]); + assert_eq!(crate::ocr::inflate_capped(&payload, 1000), None); + // At exactly the required cap the same stream inflates fine. + let ok = crate::ocr::inflate_capped(&payload, 10 * 1024 * 1024).expect("fits the cap"); + assert_eq!(ok.len(), 10 * 1024 * 1024); + } + + #[test] + fn extract_pdf_image_dict_declaring_oversize_dims_is_skipped_pre_decode() { + // MAX_OCR_PIXELS in the PDF lane: dict-declared 100k x 100k is + // refused before any sample is touched; the page contributes nothing. + let bytes = fixtures::pdf_with_image_xobject( + "/Width 100000 /Height 100000 /ColorSpace /DeviceRGB /BitsPerComponent 8", + b"tiny", + ); + let f = expect_failed(extract_with_ocr( + &bytes, + Some("huge.pdf"), + MAX_EXTRACT_CHARS, + &PanicOcr, + )); + assert_eq!(f, ExtractFailure::ScannedPdf); + } + + #[test] + fn extract_image_lane_rejects_lying_jpeg_dims_before_decode() { + // MAX_OCR_PIXELS in the standalone-image lane: the JPEG header claims + // 65500 x 65500 (~4.3 gigapixels); the header check must refuse it + // BEFORE any pixel decode, engine provably untouched. + let bytes = fixtures::jpeg_with_lying_dims(8, 8, 65500, 65500); + let f = expect_failed(extract_with_ocr( + &bytes, + Some("liar.jpg"), + MAX_EXTRACT_CHARS, + &PanicOcr, + )); + match f { + ExtractFailure::ImageParse(msg) => assert!( + msg.contains("refuses images over"), + "must be the pre-decode dimension refusal, got: {msg}" + ), + other => panic!("expected ImageParse, got {other:?}"), + } + } + + #[test] + fn extract_pdf_dct_image_with_lying_header_dims_is_rejected_pre_decode() { + // Same lie inside a PDF: the dict claims 8x8 (passes), but the JPEG's + // own header claims gigapixels — the header re-check must catch it + // before load, and the page then contributes nothing (typed). + let jpeg = fixtures::jpeg_with_lying_dims(8, 8, 65500, 65500); + let bytes = fixtures::pdf_with_image_xobject( + "/Width 8 /Height 8 /ColorSpace /DeviceRGB /BitsPerComponent 8 /Filter /DCTDecode", + &jpeg, + ); + let f = expect_failed(extract_with_ocr( + &bytes, + Some("liar.pdf"), + MAX_EXTRACT_CHARS, + &PanicOcr, + )); + assert_eq!(f, ExtractFailure::ScannedPdf); + } + + #[test] + fn extract_pdf_flate_wrapped_dct_image_decodes_and_ocrs() { + // [/FlateDecode /DCTDecode]: previously dead (lopdf errors + // Unimplemented on DCT); now the flate layer is stripped manually + // (capped) and the JPEG rides the normal bomb-checked decode. + let jpeg = fixtures::jpeg_bytes(24, 16); + let bytes = fixtures::pdf_with_image_xobject( + "/Width 24 /Height 16 /ColorSpace /DeviceRGB /BitsPerComponent 8 \ + /Filter [/FlateDecode /DCTDecode]", + &fixtures::zlib(&jpeg), + ); + let ex = expect_extracted(extract_with_ocr( + &bytes, + Some("wrapped.pdf"), + MAX_EXTRACT_CHARS, + &FakeOcr("the falcon codeword is zanzibar"), + )); + assert_eq!(ex.method, "pdf-ocr"); + assert_eq!( + ex.ocr_pages, + Some(crate::ocr::OcrPages { + ocred: 1, + total: 1, + skipped_unsupported: 0 + }) + ); + assert!(ex.text.contains("the falcon codeword is zanzibar")); + } + + #[test] + fn extract_pdf_all_pages_unsupported_encoding_fails_with_the_distinct_reason() { + // A CCITT-only scan: OCR never gets to ATTEMPT anything, and saying + // "OCR found none" would be a false claim of having looked (C1). + let bytes = fixtures::pdf_with_image_xobject( + "/Width 8 /Height 8 /ColorSpace /DeviceGray /BitsPerComponent 1 \ + /Filter /CCITTFaxDecode", + b"not really ccitt data", + ); + let f = expect_failed(extract_with_ocr( + &bytes, + Some("fax.pdf"), + MAX_EXTRACT_CHARS, + &PanicOcr, + )); + assert_eq!(f, ExtractFailure::UnsupportedPdfImages); + assert_eq!( + f.reason(), + "scanned/image PDF — unsupported image encodings; OCR could not attempt" + ); + // The accounting the receipt would carry on a partial success: + let mut budget = Budget::new(MAX_EXTRACT_CHARS); + let pages = + crate::ocr::ocr_pdf_pages(&bytes, &mut budget, crate::ocr::MAX_OCR_PAGES, &PanicOcr) + .expect("walk parses"); + assert_eq!( + pages, + crate::ocr::OcrPages { + ocred: 0, + total: 1, + skipped_unsupported: 1 + } + ); + } + + #[test] + fn extract_pdf_flate_image_with_png_predictor_reconstructs_exact_pixels() { + // The predictor handling lopdf used to apply inside its (uncapped) + // inflate, proven preserved on the capped path: Sub- and Up-filtered + // rows must reconstruct to the exact original samples. + let raw_rows: [[u8; 4]; 2] = [[10, 20, 30, 40], [50, 60, 70, 80]]; + let filtered = [ + [1u8, 10, 10, 10, 10], // Sub: first byte raw, then deltas of 10 + [2u8, 40, 40, 40, 40], // Up: deltas against the row above + ] + .concat(); + let bytes = fixtures::pdf_with_image_xobject( + "/Width 4 /Height 2 /ColorSpace /DeviceGray /BitsPerComponent 8 \ + /Filter /FlateDecode \ + /DecodeParms << /Predictor 15 /Colors 1 /BitsPerComponent 8 /Columns 4 >>", + &fixtures::zlib(&filtered), + ); + let capture = CaptureOcr::new(); + let ex = expect_extracted(extract_with_ocr( + &bytes, + Some("predicted.pdf"), + MAX_EXTRACT_CHARS, + &capture, + )); + assert_eq!(ex.method, "pdf-ocr"); + let seen = capture.0.lock().unwrap(); + assert_eq!(seen.len(), 1, "exactly one bitmap must reach the engine"); + let (w, h, rgb) = &seen[0]; + assert_eq!((*w, *h), (4, 2)); + let expected: Vec = raw_rows.iter().flatten().flat_map(|&g| [g, g, g]).collect(); + assert_eq!(rgb, &expected, "unfiltered gray samples, replicated to RGB"); + } + + #[test] + fn extract_pdf_plain_flate_rgb_image_still_decodes() { + // The legit raw-samples lane must survive the bomb-guard rewrite. + let samples: Vec = (0..2u8 * 2 * 3).map(|i| i * 10).collect(); + let bytes = fixtures::pdf_with_image_xobject( + "/Width 2 /Height 2 /ColorSpace /DeviceRGB /BitsPerComponent 8 /Filter /FlateDecode", + &fixtures::zlib(&samples), + ); + let capture = CaptureOcr::new(); + let ex = expect_extracted(extract_with_ocr( + &bytes, + Some("raw.pdf"), + MAX_EXTRACT_CHARS, + &capture, + )); + assert_eq!(ex.method, "pdf-ocr"); + let seen = capture.0.lock().unwrap(); + assert_eq!(seen.len(), 1); + assert_eq!(seen[0].2, samples); + } + + #[test] + fn receipt_json_discloses_ocr_page_accounting_only_when_present() { + let r = receipt_json( + "pdf-ocr", + false, + Some(crate::ocr::OcrPages { + ocred: 3, + total: 7, + skipped_unsupported: 2, + }), + ); + assert_eq!(r["method"], "pdf-ocr"); + assert_eq!(r["pages_ocred"], 3); + assert_eq!(r["pages_total"], 7); + assert_eq!(r["pages_skipped_unsupported"], 2); + let r = receipt_json("pdf-text", true, None); + assert_eq!(r["truncated"], true); + for key in ["pages_ocred", "pages_total", "pages_skipped_unsupported"] { + assert!( + r.get(key).is_none(), + "non-OCR receipts must not carry {key}" + ); + } + } + + #[test] + fn png_and_jpeg_extract_via_image_ocr() { + for (bytes, name) in [ + (fixtures::png_bytes(20, 12), "shot.png"), + (fixtures::jpeg_bytes(20, 12), "photo.jpg"), + ] { + let ex = expect_extracted(extract_with_ocr( + &bytes, + Some(name), + MAX_EXTRACT_CHARS, + &FakeOcr("renewal quote is 61000"), + )); + assert_eq!(ex.method, "image-ocr", "for {name}"); + assert_eq!(ex.ocr_pages, None); + assert!(ex.text.contains("renewal quote is 61000")); + } + } + + #[test] + fn image_detected_by_magic_even_with_misleading_name() { + // Magic wins: PNG bytes named .pdf still ride the image-ocr path. + let ex = expect_extracted(extract_with_ocr( + &fixtures::png_bytes(20, 12), + Some("report.pdf"), + MAX_EXTRACT_CHARS, + &FakeOcr("hello"), + )); + assert_eq!(ex.method, "image-ocr"); + } + + #[test] + fn corrupt_image_is_a_typed_image_parse_failure() { + // Valid PNG magic, hostile body. + let mut bytes = b"\x89PNG\r\n\x1a\n".to_vec(); + bytes.extend_from_slice(&[0xAB; 128]); + let f = expect_failed(extract_with_ocr( + &bytes, + Some("bad.png"), + MAX_EXTRACT_CHARS, + &PanicOcr, // undecodable ⇒ the engine is never consulted + )); + assert!(matches!(f, ExtractFailure::ImageParse(_)), "got {f:?}"); + } + + #[test] + fn image_where_ocr_finds_nothing_is_a_typed_no_text_failure() { + let f = expect_failed(extract_with_ocr( + &fixtures::png_bytes(20, 12), + Some("blank.png"), + MAX_EXTRACT_CHARS, + &FakeOcr(""), + )); + assert_eq!(f, ExtractFailure::ImageNoText); + assert_eq!(f.reason(), "image parsed but OCR found no text"); + } + + #[test] + fn image_with_ocr_engine_down_is_a_typed_ocr_unavailable_failure() { + let f = expect_failed(extract_with_ocr( + &fixtures::png_bytes(20, 12), + Some("shot.png"), + MAX_EXTRACT_CHARS, + &DownOcr, + )); + assert!(matches!(f, ExtractFailure::OcrUnavailable(_)), "got {f:?}"); + } + + /// End-to-end with the REAL engine (downloads ~12 MB of models on first + /// run): gated behind VERITY_OCR_E2E=1. Proves the full lane — rendered + /// text PNG → image-ocr, and the same image embedded as a scanned PDF → + /// pdf-ocr — recognizes the known sentence. + #[test] + fn e2e_real_engine_reads_rendered_text_from_png_and_scanned_pdf() { + if std::env::var("VERITY_OCR_E2E").as_deref() != Ok("1") { + eprintln!("VERITY_OCR_E2E != 1; skipping"); + return; + } + let png = include_bytes!("testdata/ocr-sample.png"); + let ex = expect_extracted(extract(png, Some("ocr-sample.png"))); + assert_eq!(ex.method, "image-ocr"); + let lower = ex.text.to_lowercase(); + assert!( + lower.contains("quick brown fox"), + "OCR text was: {:?}", + ex.text + ); + + let jpeg = include_bytes!("testdata/ocr-sample.jpg"); + let pdf = fixtures::scanned_pdf_with_jpegs(&[jpeg.as_slice()]); + let ex = expect_extracted(extract(&pdf, Some("scan.pdf"))); + assert_eq!(ex.method, "pdf-ocr"); + assert_eq!( + ex.ocr_pages, + Some(crate::ocr::OcrPages { + ocred: 1, + total: 1, + skipped_unsupported: 0 + }) + ); + let lower = ex.text.to_lowercase(); + assert!( + lower.contains("quick brown fox"), + "OCR text was: {:?}", + ex.text ); } @@ -1183,10 +2049,19 @@ mod tests { extract(b"hello world", Some("notes.txt")), ExtractOutcome::NotHandled )); + // Unknown binary with no format claim: not our job. assert!(matches!( - extract(&[0x89, b'P', b'N', b'G', 0x0d, 0x0a], Some("img.png")), + extract(&[0x00, 0x01, 0x02, 0x03, 0x7f], Some("blob.bin")), ExtractOutcome::NotHandled )); + // A TRUNCATED png magic under a .png name: the bytes lie about the + // claim (magic wins), so this is now a typed refusal, not a silent + // store-only — PNG became one of our formats with the OCR tier. + let f = expect_failed(extract( + &[0x89, b'P', b'N', b'G', 0x0d, 0x0a], + Some("img.png"), + )); + assert_eq!(f, ExtractFailure::UnrecognizedFormat); } #[test] diff --git a/crates/verity-server/src/main.rs b/crates/verity-server/src/main.rs index b5c5c10..6e33028 100644 --- a/crates/verity-server/src/main.rs +++ b/crates/verity-server/src/main.rs @@ -43,6 +43,7 @@ mod media; #[cfg(test)] mod media_tests; mod metrics; +mod ocr; mod playground; #[cfg(test)] mod principals_tests; @@ -3830,7 +3831,8 @@ struct IngestDocumentsRequest { #[serde(default)] content: Option, /// Binary path (Tier-1 extraction, extract.rs): raw file bytes, base64. - /// The SERVER extracts text (PDF/PPTX/XLS(X), deterministic, no OCR) so + /// The SERVER extracts text (PDF/PPTX/XLS(X)/DOC(X) text layers, plus + /// local best-effort OCR for scanned PDFs and PNG/JPEG — ocr.rs) so /// connectors stay extraction-free. Chosen over posting to /v1/files /// because /v1/files stamps the uploader scope's principals — it would /// REPLACE the connector's mirrored per-item ACL, which is the whole @@ -4274,11 +4276,17 @@ pub(crate) async fn ingest_document( (Some(c), hash) } DeliveredContent::Bytes { raw, hash_over } => { - let text = match extract::extract(&raw, req.filename.as_deref()) { + // Extraction runs on the blocking pool: the OCR paths (scanned + // PDFs, images) can take seconds and must not stall the runtime. + let fname = req.filename.clone(); + let outcome = + tokio::task::spawn_blocking(move || extract::extract(&raw, fname.as_deref())) + .await + .map_err(internal)?; + let text = match outcome { extract::ExtractOutcome::Extracted(ex) => { - extraction_receipt = Some(serde_json::json!({ - "method": ex.method, "truncated": ex.truncated, - })); + extraction_receipt = + Some(extract::receipt_json(ex.method, ex.truncated, ex.ocr_pages)); Some(ex.text) } extract::ExtractOutcome::Failed(f) => { diff --git a/crates/verity-server/src/media.rs b/crates/verity-server/src/media.rs index 007d1ad..4c3f691 100644 --- a/crates/verity-server/src/media.rs +++ b/crates/verity-server/src/media.rs @@ -1,12 +1,14 @@ //! MediaObject + signed URIs (roadmap task 9): blobs live in the `media` //! table, addressed by uuid, served ONLY through HMAC-signed, expiring URLs //! minted under a scope handle. Text-like media additionally chunks into the -//! retrieval index under the uploader's scope; PDF / PPTX / XLS(X) go through -//! the Tier-1 extractor (extract.rs — deterministic, Rust-native, no OCR) and -//! index the extracted text, with the method + truncation recorded in -//! provenance and typed extraction failures stored metadata-only, disclosed -//! in both the response and the episode record. Other binary media is -//! store-only. +//! retrieval index under the uploader's scope; PDF / PPTX / XLS(X) / DOC(X) / +//! PNG / JPEG go through the Tier-1 extractor (extract.rs — Rust-native and +//! local; scanned PDFs and images ride the best-effort local OCR tier, ocr.rs) +//! and index the extracted text, with the method + truncation (+ OCR page +//! accounting for OCRed PDFs) recorded in provenance and typed extraction +//! failures stored +//! metadata-only, disclosed in both the response and the episode record. Other +//! binary media is store-only. //! //! v0.2 seam, stated honestly: the signed GET enforces signature + expiry //! (and the sign step enforces the tenant match), but per-principal media @@ -287,15 +289,28 @@ pub(crate) async fn ingest_file( text: String, method: &'static str, truncated: bool, + ocr_pages: Option, }, Refuse(crate::extract::ExtractFailure), StoreOnly, } - let plan = match crate::extract::extract(&bytes, filename.as_deref()) { + // Extraction runs on the blocking pool: the OCR paths can take seconds + // per page, which must never stall the async runtime. + let (outcome, bytes) = { + let fname = filename.clone(); + tokio::task::spawn_blocking(move || { + let outcome = crate::extract::extract(&bytes, fname.as_deref()); + (outcome, bytes) + }) + .await + .map_err(internal)? + }; + let plan = match outcome { crate::extract::ExtractOutcome::Extracted(ex) => Plan::Index { text: ex.text, method: ex.method, truncated: ex.truncated, + ocr_pages: ex.ocr_pages, }, crate::extract::ExtractOutcome::Failed(f) => Plan::Refuse(f), crate::extract::ExtractOutcome::NotHandled => { @@ -307,6 +322,7 @@ pub(crate) async fn ingest_file( text: s.to_string(), method: "utf-8", truncated: false, + ocr_pages: None, }, None => Plan::StoreOnly, } @@ -320,6 +336,7 @@ pub(crate) async fn ingest_file( text, method, truncated, + ocr_pages, } => { let episode_id = state .storage @@ -331,7 +348,7 @@ pub(crate) async fn ingest_file( payload: serde_json::json!({ "media_id": media_id, "filename": filename, "mime": mime, "sha256": sha256, "size_bytes": bytes.len(), - "extraction": { "method": method, "truncated": truncated }, + "extraction": crate::extract::receipt_json(method, truncated, ocr_pages), }), content_hash: sha256.clone(), trust_tier: TrustTier::Observation, @@ -374,10 +391,7 @@ pub(crate) async fn ingest_file( // emit after a write. Its absence here meant file content added via // `verity-cli add` (POST /v1/files) was never entity-resolved. state.resolution.mark_dirty(payload.tenant_id); - extraction_receipt = Some(serde_json::json!({ - "method": method, - "truncated": truncated, - })); + extraction_receipt = Some(crate::extract::receipt_json(method, truncated, ocr_pages)); } Plan::Refuse(failure) => { let reason = failure.reason(); diff --git a/crates/verity-server/src/ocr.rs b/crates/verity-server/src/ocr.rs new file mode 100644 index 0000000..bf8c45e --- /dev/null +++ b/crates/verity-server/src/ocr.rs @@ -0,0 +1,913 @@ +//! Local OCR tier: scanned PDFs and standalone PNG/JPEG images → text. +//! +//! Sovereignty-first, same honesty rules as extract.rs (which owns the policy; +//! this module is the mechanism): +//! +//! * **Pure Rust, zero system dependencies.** The engine is the `ocrs` crate +//! on the `rten` runtime — no tesseract (a system dep would break the +//! clean-VM stranger gate), no cloud OCR, no Python. OCR quality is +//! printed-text-grade and best-effort: fine for scans and screenshots of +//! type, not handwriting, and receipts always disclose the method +//! ("pdf-ocr" / "image-ocr") so a consumer can weigh the text accordingly. +//! * **Models fetch once, then cache.** The two rten models (~2.5 MB +//! detection + ~9.7 MB recognition) download on FIRST USE from the ocrs +//! project's canonical bucket into `~/.cache/ocrs` — the same directory the +//! `ocrs` CLI uses, and the same fetch-once-then-cache lane as the MiniLM +//! query encoder (which caches via hf-hub). `VERITY_OCR_MODEL_DIR` +//! overrides the location for air-gapped deployments (drop the two `.rten` +//! files there by hand). Downloads run under real connect/global timeouts, +//! are single-flighted process-wide, and every file — downloaded or cached — +//! must match a pinned SHA-256 before rten sees it (a corrupt cache +//! self-heals with one refetch). The engine is only ever initialized when +//! there is actually an image to recognize — a text PDF never touches this +//! module. +//! * **Failure is typed, never silent, never fatal.** Download/init/inference +//! failure surfaces as an `Err(String)` that extract.rs turns into the +//! disclosed `OCR unavailable: …` extraction failure. A failed init is NOT +//! cached — the next file retries (a transient network blip during the +//! first-ever scan must not poison the process). +//! * **Bounded work.** Scanned PDFs OCR at most [`MAX_OCR_PAGES`] pages; +//! individual images over [`MAX_OCR_PIXELS`] are refused (decompression- +//! bomb guard). Char caps are enforced by the caller's `Budget`. +//! +//! The `ocrs`/`rten` engine itself sits behind the default-ON `ocr` cargo +//! feature; built without it, [`default_backend`] returns a stub whose every +//! call fails typed ("built without the 'ocr' feature") — the decode and +//! PDF-walking plumbing stays compiled and tested either way. + +use std::io::Cursor; + +use crate::extract::Budget; + +/// Hard cap on pages OCRed per scanned PDF. Disclosed via the receipt's +/// `pages_ocred` (a 200-page scan reporting `pages_ocred: 50` is visibly +/// partial; the `truncated` flag additionally covers the char cap). +pub(crate) const MAX_OCR_PAGES: usize = 50; + +/// Refuse to decode images beyond this many pixels (~40 MP — comfortably +/// above any sane scan, small enough that a crafted PNG bomb cannot balloon +/// into gigabytes of raster). +pub(crate) const MAX_OCR_PIXELS: u64 = 40_000_000; + +/// The OCR seam. Implementations recognize printed text in one RGB8 bitmap +/// (`rgb.len() == 3 * width * height`). +/// +/// `Err` means the ENGINE failed (models unavailable, inference error) and is +/// disclosed as such; "ran fine, found no text" is `Ok` with an empty string. +/// Tests inject fakes here; production uses [`default_backend`]. +pub(crate) trait OcrBackend: Sync { + fn recognize_rgb(&self, width: u32, height: u32, rgb: &[u8]) -> Result; +} + +/// The production backend: lazily initialized global ocrs engine (or the +/// typed always-fails stub when built without the `ocr` feature). +pub(crate) fn default_backend() -> &'static dyn OcrBackend { + &engine::LazyOcrs +} + +// --------------------------------------------------------------------------- +// Standalone image decode (PNG/JPEG bytes → RGB8, bomb-guarded) +// --------------------------------------------------------------------------- + +/// Decode PNG/JPEG bytes to RGB8. Dimensions are read from the header FIRST +/// and checked against [`MAX_OCR_PIXELS`] before any pixel is materialized. +pub(crate) fn decode_rgb(bytes: &[u8]) -> Result { + let (w, h) = image::ImageReader::new(Cursor::new(bytes)) + .with_guessed_format() + .map_err(|e| e.to_string())? + .into_dimensions() + .map_err(|e| e.to_string())?; + check_pixels(w, h)?; + let img = image::ImageReader::new(Cursor::new(bytes)) + .with_guessed_format() + .map_err(|e| e.to_string())? + .decode() + .map_err(|e| e.to_string())?; + Ok(img.into_rgb8()) +} + +fn check_pixels(w: u32, h: u32) -> Result<(), String> { + if u64::from(w) * u64::from(h) > MAX_OCR_PIXELS { + return Err(format!( + "image is {w}x{h} pixels; OCR refuses images over {MAX_OCR_PIXELS} pixels" + )); + } + Ok(()) +} + +// --------------------------------------------------------------------------- +// Scanned-PDF page walk: embedded image XObjects, in page order +// --------------------------------------------------------------------------- + +/// Why a scanned-PDF OCR pass failed. extract.rs maps these to its typed +/// `ExtractFailure` reasons. +#[derive(Debug)] +pub(crate) enum PdfOcrError { + /// lopdf could not re-parse the document for the image walk. + Parse(String), + /// The OCR engine itself failed (model download/init/inference). + Engine(String), +} + +/// The honest page accounting for a scanned-PDF OCR pass, disclosed verbatim +/// on the extraction receipt: `ocred < total` makes a partial pass visible; +/// `skipped_unsupported` makes "we never even attempted this page" visible. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct OcrPages { + /// Pages where at least one image decoded and the engine was consulted. + pub(crate) ocred: u32, + /// Total pages in the document (NOT capped at [`MAX_OCR_PAGES`] — the + /// receipt must show how much of the document the pass could ever cover). + pub(crate) total: u32, + /// Walked pages that carried image XObjects, none of which we could + /// decode because every one used an encoding we don't implement + /// (CCITT/JBIG2/JPX, exotic colorspaces/bit depths). OCR never attempted + /// these pages; the caller reports that distinctly from "OCR found none". + pub(crate) skipped_unsupported: u32, +} + +/// Walk a (text-layer-less) PDF's pages in order, decode each page's embedded +/// raster image XObjects, OCR them, and push `Page N:` blocks into `budget`. +/// +/// Returns the per-page accounting. Undecodable images (bombs, lying +/// dimensions, corrupt data) and unsupported-encoding images (CCITT/JBIG2/ +/// JPX) are skipped and counted — pages built from them contribute nothing, +/// and if NOTHING contributes, the caller declares the honest typed failure. +/// The engine is never consulted when no image decodes, so a blank-page PDF +/// stays cheap and a broken model cache is only ever reported on files that +/// needed it. +pub(crate) fn ocr_pdf_pages( + bytes: &[u8], + budget: &mut Budget, + max_pages: usize, + ocr: &dyn OcrBackend, +) -> Result { + let doc = lopdf::Document::load_mem(bytes).map_err(|e| PdfOcrError::Parse(e.to_string()))?; + let all_pages = doc.get_pages(); + let mut pages = OcrPages { + ocred: 0, + total: u32::try_from(all_pages.len()).unwrap_or(u32::MAX), + skipped_unsupported: 0, + }; + for (page_no, page_id) in all_pages.into_iter().take(max_pages) { + let scan = page_images(&doc, page_id); + let page_had_image = !scan.images.is_empty(); + let mut page_text = String::new(); + for img in scan.images { + let text = ocr + .recognize_rgb(img.width(), img.height(), img.as_raw()) + .map_err(PdfOcrError::Engine)?; + let text = text.trim(); + if !text.is_empty() { + if !page_text.is_empty() { + page_text.push('\n'); + } + page_text.push_str(text); + } + } + if page_had_image { + pages.ocred += 1; + } else if scan.skipped_unsupported > 0 { + pages.skipped_unsupported += 1; + } + if !page_text.is_empty() && !budget.push(&format!("Page {page_no}:\n{page_text}\n\n")) { + break; // char cap reached — disclosed via `truncated` + } + } + Ok(pages) +} + +/// One page's decode results: the images the engine will see, plus the counts +/// of what was skipped (and why) for the honest page accounting above. +struct PageScan { + images: Vec, + /// Encodings we don't implement (CCITT/JBIG2/JPX, exotic colorspaces…). + skipped_unsupported: u32, + /// Bad data: declared-oversize dims, flate payloads over the size cap + /// (bomb guard), truncated samples, corrupt JPEG bytes. + skipped_undecodable: u32, +} + +/// What [`decode_image_xobject`] decided about one image stream. +enum XObjectImage { + Decoded(image::RgbImage), + Unsupported, + Undecodable, +} + +/// Decode the raster image XObjects a page references, in the order the +/// resource dictionary lists them. Only self-describing or raw formats we can +/// reconstruct deterministically are attempted: +/// +/// * `DCTDecode` — the stream IS a JPEG (possibly flate-wrapped); hand it to +/// the image crate after a header-dimensions bomb check. +/// * `FlateDecode` / unfiltered — raw samples; reconstructed for 8-bit +/// DeviceRGB and DeviceGray, inflated under a hard cap. +/// +/// Anything else (CCITT, JBIG2, JPX, indexed palettes, exotic bit depths) is +/// skipped and counted, never guessed at. +fn page_images(doc: &lopdf::Document, page_id: lopdf::ObjectId) -> PageScan { + let mut out = PageScan { + images: Vec::new(), + skipped_unsupported: 0, + skipped_undecodable: 0, + }; + let Ok((inline_dict, resource_ids)) = doc.get_page_resources(page_id) else { + return out; + }; + let mut resource_dicts: Vec<&lopdf::Dictionary> = inline_dict.into_iter().collect(); + for id in resource_ids { + if let Ok(d) = doc.get_dictionary(id) { + resource_dicts.push(d); + } + } + for resources in resource_dicts { + let xobjects = match resources.get(b"XObject") { + Ok(lopdf::Object::Dictionary(d)) => d, + Ok(lopdf::Object::Reference(id)) => match doc.get_dictionary(*id) { + Ok(d) => d, + Err(_) => continue, + }, + _ => continue, + }; + for (_name, obj) in xobjects.iter() { + let stream = match obj { + lopdf::Object::Reference(id) => { + match doc.get_object(*id).and_then(lopdf::Object::as_stream) { + Ok(s) => s, + Err(_) => continue, + } + } + lopdf::Object::Stream(s) => s, + _ => continue, + }; + let subtype = stream.dict.get(b"Subtype").and_then(lopdf::Object::as_name); + if !matches!(subtype, Ok(b"Image")) { + continue; + } + match decode_image_xobject(stream) { + XObjectImage::Decoded(img) => out.images.push(img), + XObjectImage::Unsupported => out.skipped_unsupported += 1, + XObjectImage::Undecodable => out.skipped_undecodable += 1, + } + } + } + out +} + +/// Slack allowed on top of the exact expected sample size when inflating a +/// FlateDecode image stream: PNG predictors prepend 1 filter byte per row +/// (covered by `+ height` at the call sites) and this small fixed margin +/// absorbs encoder padding. Anything beyond the cap is a bomb — skipped. +const INFLATE_MARGIN: usize = 1024; + +/// Inflate a FlateDecode payload with a hard output cap. This exists because +/// lopdf's `decompressed_content()` inflates with NO output bound, so a +/// ~100 KB stream declaring a 10x10 image can materialize gigabytes and OOM +/// the server. Memory here is bounded at `cap + 1` bytes: the decoder is read +/// through `Read::take(cap + 1)`, and landing past `cap` means the stream +/// lied about its size → `None` (the caller skips the image, counted). +/// +/// Mirrors lopdf's decode quirks so behavior on legit files is unchanged: +/// a zlib failure that produced no output retries as raw deflate (corrupt +/// zlib headers/checksums in the wild), and a mid-stream error keeps the +/// partial output — the caller's expected-length check decides its fate. +pub(crate) fn inflate_capped(input: &[u8], cap: usize) -> Option> { + use std::io::Read as _; + + let limit = cap as u64 + 1; // one probe byte past the cap detects overflow + let mut out = Vec::new(); + let result = flate2::read::ZlibDecoder::new(input) + .take(limit) + .read_to_end(&mut out); + if result.is_err() && out.is_empty() && input.len() > 2 { + let _ = flate2::read::DeflateDecoder::new(&input[2..]) + .take(limit) + .read_to_end(&mut out); + } + if out.len() > cap { + return None; + } + Some(out) +} + +/// Undo PNG predictors (10-15) per the stream's `/DecodeParms`, row by row — +/// ported from the lopdf path we no longer take (its `filters::png` unfilter +/// ran inside the uncapped `decompressed_content()`). Predictor 1/2 and +/// absent params pass the data through untouched, exactly as lopdf did. +/// `None` means malformed predictor data (bad filter tag, ragged final row). +pub(crate) fn png_unpredict(data: Vec, params: Option<&lopdf::Dictionary>) -> Option> { + let Some(params) = params else { + return Some(data); + }; + let get = |key: &[u8]| params.get(key).and_then(lopdf::Object::as_i64).ok(); + let predictor = get(b"Predictor").unwrap_or(1); + if !(10..=15).contains(&predictor) { + return Some(data); + } + let columns = get(b"Columns").unwrap_or(1).max(1) as usize; + let colors = get(b"Colors").unwrap_or(1).max(1) as usize; + let bits = get(b"BitsPerComponent").unwrap_or(8).max(8) as usize; + let bpp = colors * bits / 8; + let row_len = bpp.checked_mul(columns)?; + let stride = row_len.checked_add(1)?; // +1 leading filter-type byte + if row_len == 0 || !data.len().is_multiple_of(stride) { + return None; + } + let mut prev = vec![0u8; row_len]; + let mut out = Vec::with_capacity(data.len() / stride * row_len); + for chunk in data.chunks_exact(stride) { + let mut cur = chunk[1..].to_vec(); + png_unfilter_row(chunk[0], bpp, &prev, &mut cur)?; + out.extend_from_slice(&cur); + prev = cur; + } + Some(out) +} + +/// One row of PNG unfiltering (RFC 2083 §6): None/Sub/Up/Average/Paeth. +fn png_unfilter_row(filter: u8, bpp: usize, prev: &[u8], cur: &mut [u8]) -> Option<()> { + let bpp = bpp.min(cur.len()); + match filter { + 0 => {} + 1 => { + for i in bpp..cur.len() { + cur[i] = cur[i].wrapping_add(cur[i - bpp]); + } + } + 2 => { + for i in 0..cur.len() { + cur[i] = cur[i].wrapping_add(prev[i]); + } + } + 3 => { + for i in 0..bpp { + cur[i] = cur[i].wrapping_add(prev[i] / 2); + } + for i in bpp..cur.len() { + let avg = (u16::from(cur[i - bpp]) + u16::from(prev[i])) / 2; + cur[i] = cur[i].wrapping_add(avg as u8); + } + } + 4 => { + for i in 0..bpp { + cur[i] = cur[i].wrapping_add(paeth_predict(0, prev[i], 0)); + } + for i in bpp..cur.len() { + cur[i] = cur[i].wrapping_add(paeth_predict(cur[i - bpp], prev[i], prev[i - bpp])); + } + } + _ => return None, + } + Some(()) +} + +fn paeth_predict(left: u8, above: u8, upper_left: u8) -> u8 { + let (l, a, ul) = (i16::from(left), i16::from(above), i16::from(upper_left)); + let estimate = l + a - ul; + let (dl, da, dul) = ( + (estimate - l).abs(), + (estimate - a).abs(), + (estimate - ul).abs(), + ); + if dl <= da && dl <= dul { + left + } else if da <= dul { + above + } else { + upper_left + } +} + +fn decode_image_xobject(stream: &lopdf::Stream) -> XObjectImage { + let dict = &stream.dict; + let dim = |key: &[u8]| { + dict.get(key) + .and_then(lopdf::Object::as_i64) + .ok() + .and_then(|v| u32::try_from(v).ok()) + }; + let (Some(width), Some(height)) = (dim(b"Width"), dim(b"Height")) else { + return XObjectImage::Undecodable; + }; + if check_pixels(width, height).is_err() { + return XObjectImage::Undecodable; + } + + // Filter may be a single name, an array, or absent (raw samples). + let filters: Vec<&[u8]> = match dict.get(b"Filter") { + Ok(lopdf::Object::Name(n)) => vec![n.as_slice()], + Ok(lopdf::Object::Array(a)) => a.iter().filter_map(|o| o.as_name().ok()).collect(), + _ => vec![], + }; + let params = dict + .get(b"DecodeParms") + .and_then(lopdf::Object::as_dict) + .ok(); + + if filters.last() == Some(&b"DCTDecode".as_slice()) { + // The stream content is a JPEG file, possibly flate-wrapped. lopdf's + // decompressed_content() errors Unimplemented on DCT streams, so the + // flate layer is stripped manually — capped: a real JPEG payload for + // a dict-declared-legal image fits well under raw RGB size. + let jpeg: std::borrow::Cow<'_, [u8]> = match filters.as_slice() { + [_] => std::borrow::Cow::Borrowed(&stream.content), + [f, _] if *f == b"FlateDecode" => { + let cap = match (width as usize) + .checked_mul(height as usize) + .and_then(|p| p.checked_mul(3)) + .and_then(|p| p.checked_add(INFLATE_MARGIN)) + { + Some(cap) => cap, + None => return XObjectImage::Undecodable, + }; + match inflate_capped(&stream.content, cap) { + Some(jpeg) => std::borrow::Cow::Owned(jpeg), + None => return XObjectImage::Undecodable, + } + } + _ => return XObjectImage::Unsupported, + }; + // decode_rgb reads the JPEG's OWN header dimensions and enforces + // MAX_OCR_PIXELS BEFORE any pixel decodes — the dict check above only + // covered the *claimed* size, and headers can lie. + return match decode_rgb(&jpeg) { + Ok(img) => XObjectImage::Decoded(img), + Err(_) => XObjectImage::Undecodable, + }; + } + + // Raw samples (optionally flate-compressed): 8-bit DeviceRGB/DeviceGray. + let bpc = dict + .get(b"BitsPerComponent") + .and_then(lopdf::Object::as_i64); + if !matches!(bpc, Ok(8)) { + return XObjectImage::Unsupported; + } + let channels: usize = match dict.get(b"ColorSpace").and_then(lopdf::Object::as_name) { + Ok(b"DeviceRGB") => 3, + Ok(b"DeviceGray") => 1, + _ => return XObjectImage::Unsupported, + }; + let expected = match (width as usize) + .checked_mul(height as usize) + .and_then(|p| p.checked_mul(channels)) + { + Some(e) => e, + None => return XObjectImage::Undecodable, + }; + let raw = if filters.is_empty() { + stream.content.clone() + } else if filters == [b"FlateDecode".as_slice()] { + // Capped inflation (never lopdf's uncapped decompressed_content): + // expected samples + 1 predictor filter byte per row + fixed margin. + // A stream inflating past that declared-size envelope is a bomb. + let cap = expected + .saturating_add(height as usize) + .saturating_add(INFLATE_MARGIN); + let inflated = match inflate_capped(&stream.content, cap) { + Some(data) => data, + None => return XObjectImage::Undecodable, + }; + match png_unpredict(inflated, params) { + Some(data) => data, + None => return XObjectImage::Undecodable, + } + } else { + return XObjectImage::Unsupported; + }; + if raw.len() < expected { + return XObjectImage::Undecodable; + } + let img = match channels { + 3 => image::RgbImage::from_raw(width, height, raw[..expected].to_vec()), + 1 => { + let rgb: Vec = raw[..expected].iter().flat_map(|&g| [g, g, g]).collect(); + image::RgbImage::from_raw(width, height, rgb) + } + _ => unreachable!(), + }; + match img { + Some(img) => XObjectImage::Decoded(img), + None => XObjectImage::Undecodable, + } +} + +// --------------------------------------------------------------------------- +// The engine (feature "ocr"): lazy global ocrs instance + model fetch +// --------------------------------------------------------------------------- + +#[cfg(feature = "ocr")] +mod engine { + use std::path::{Path, PathBuf}; + use std::sync::atomic::{AtomicU64, Ordering}; + use std::sync::{Mutex, OnceLock}; + use std::time::Duration; + + use super::OcrBackend; + + /// Canonical distribution point for the ocrs models (the same URLs the + /// `ocrs` CLI fetches; the author's Hugging Face repo only carries the + /// pre-2024 model format, which rten ≥0.16 no longer loads). + const DETECTION_MODEL_URL: &str = + "https://ocrs-models.s3-accelerate.amazonaws.com/text-detection.rten"; + const RECOGNITION_MODEL_URL: &str = + "https://ocrs-models.s3-accelerate.amazonaws.com/text-recognition.rten"; + + /// Pinned SHA-256 of the canonical model artifacts above (computed from + /// the upstream files; the bucket serves stable, versioned bytes). + /// Verified after every download AND on first cache load each process — + /// a corrupt, truncated, or tampered file never reaches rten, and a bad + /// cached copy self-heals (delete + one refetch) instead of failing OCR + /// until someone clears ~/.cache/ocrs by hand. + const DETECTION_MODEL_SHA256: &str = + "f15cfb56bd02c4bf478a20343986504a1f01e1665c2b3a0ad66340f054b1b5ca"; + const RECOGNITION_MODEL_SHA256: &str = + "e484866d4cce403175bd8d00b128feb08ab42e208de30e42cd9889d8f1735a6e"; + + /// Real network bounds on the model download: without them a hung + /// connection parks the blocking extraction thread indefinitely. + const CONNECT_TIMEOUT: Duration = Duration::from_secs(10); + const GLOBAL_TIMEOUT: Duration = Duration::from_secs(120); + + /// Process-wide single-flight for the model fetch: N concurrent first + /// scans must produce ONE download, not N racing ones. + static FETCH_LOCK: Mutex<()> = Mutex::new(()); + /// Uniquifies temp download names within the process (the name also + /// carries the pid), so two healing threads can never tear each other's + /// partial files. + static TMP_SEQ: AtomicU64 = AtomicU64::new(0); + + /// Injectable fetch seam: production passes [`download`]; tests inject + /// counting/faulty fetchers to prove single-flight and self-heal without + /// touching the network. + type FetchFn<'a> = &'a (dyn Fn(&str, &Path) -> Result<(), String> + Sync); + + /// `VERITY_OCR_MODEL_DIR` override, else `~/.cache/ocrs` (shared with the + /// ocrs CLI's own cache, so nothing downloads twice on a dev machine). + fn model_dir() -> Result { + if let Some(dir) = std::env::var_os("VERITY_OCR_MODEL_DIR") { + return Ok(PathBuf::from(dir)); + } + #[cfg(windows)] + let home = std::env::var_os("USERPROFILE"); + #[cfg(not(windows))] + let home = std::env::var_os("HOME"); + let home = home.ok_or("no home directory; set VERITY_OCR_MODEL_DIR")?; + Ok(PathBuf::from(home).join(".cache").join("ocrs")) + } + + /// Download `url` to `dest` with real timeouts. Unique temp-file + rename + /// so a torn download never lands as a "cached" model. + fn download(url: &str, dest: &Path) -> Result<(), String> { + let dir = dest + .parent() + .ok_or_else(|| format!("no parent dir for {}", dest.display()))?; + std::fs::create_dir_all(dir).map_err(|e| format!("creating {}: {e}", dir.display()))?; + let agent = ureq::Agent::new_with_config( + ureq::Agent::config_builder() + .timeout_connect(Some(CONNECT_TIMEOUT)) + .timeout_global(Some(GLOBAL_TIMEOUT)) + .build(), + ); + let mut resp = agent + .get(url) + .call() + .map_err(|e| format!("downloading {url}: {e}"))?; + let tmp = dest.with_extension(format!( + "part.{}.{}", + std::process::id(), + TMP_SEQ.fetch_add(1, Ordering::Relaxed) + )); + let result = (|| -> Result<(), String> { + let mut file = std::fs::File::create(&tmp) + .map_err(|e| format!("creating {}: {e}", tmp.display()))?; + std::io::copy(&mut resp.body_mut().as_reader(), &mut file) + .map_err(|e| format!("writing {}: {e}", tmp.display()))?; + std::fs::rename(&tmp, dest) + .map_err(|e| format!("moving model into place at {}: {e}", dest.display())) + })(); + if result.is_err() { + let _ = std::fs::remove_file(&tmp); + } + result + } + + fn sha256_hex(path: &Path) -> Result { + use sha2::Digest as _; + let mut file = + std::fs::File::open(path).map_err(|e| format!("opening {}: {e}", path.display()))?; + let mut hasher = sha2::Sha256::new(); + std::io::copy(&mut file, &mut hasher) + .map_err(|e| format!("hashing {}: {e}", path.display()))?; + Ok(format!("{:x}", hasher.finalize())) + } + + fn verify_sha256(path: &Path, expected: &str) -> Result<(), String> { + let got = sha256_hex(path)?; + if got != expected { + return Err(format!( + "model checksum mismatch at {}: expected sha256 {expected}, got {got}", + path.display() + )); + } + Ok(()) + } + + /// Make `dest` present AND checksum-valid, fetching or self-healing as + /// needed. Fail-closed at every exit: a file that doesn't hash to the + /// pinned value is deleted, refetched at most ONCE, and if still wrong, + /// deleted again and reported typed — never handed to rten. + fn ensure_model_file( + url: &str, + dest: &Path, + expected_sha256: &str, + fetch: FetchFn<'_>, + ) -> Result<(), String> { + if dest.exists() { + match verify_sha256(dest, expected_sha256) { + Ok(()) => return Ok(()), + Err(_) => { + // Corrupt cache (torn pre-hardening download, disk rot, + // tampering): self-heal with one refetch. + std::fs::remove_file(dest) + .map_err(|e| format!("removing corrupt model {}: {e}", dest.display()))?; + } + } + } + fetch(url, dest)?; + verify_sha256(dest, expected_sha256).inspect_err(|_| { + let _ = std::fs::remove_file(dest); + }) + } + + /// Fetch/verify both model files under the process-wide single-flight + /// lock. Split from [`global`] so tests can prove that two racing threads + /// produce exactly one fetch. + fn ensure_models_locked( + specs: &[(&str, &Path, &str)], + fetch: FetchFn<'_>, + ) -> Result<(), String> { + let _guard = FETCH_LOCK.lock().unwrap_or_else(|p| p.into_inner()); + for (url, dest, sha) in specs { + ensure_model_file(url, dest, sha, fetch)?; + } + Ok(()) + } + + /// Load a (present, checksum-valid) model file; if loading fails anyway + /// (an artifact rten can't parse), delete it, refetch ONCE, and retry — + /// then fail typed. Generic over the loader so tests can prove the heal + /// without real rten models. + fn load_model_healing( + url: &str, + dest: &Path, + expected_sha256: &str, + fetch: FetchFn<'_>, + load: &dyn Fn(&Path) -> Result, + ) -> Result { + match load(dest) { + Ok(model) => Ok(model), + Err(first) => { + let _ = std::fs::remove_file(dest); + ensure_model_file(url, dest, expected_sha256, fetch)?; + load(dest).map_err(|e| { + format!( + "loading {} failed even after refetch: {e} (first attempt: {first})", + dest.display() + ) + }) + } + } + } + + static ENGINE: OnceLock = OnceLock::new(); + + /// Fetch-if-needed, verify, load, and cache the engine. Success is cached + /// for the process lifetime; failure is NOT (the next file retries, so a + /// transient blip during the first-ever scan must not poison the + /// process). Fetches are single-flighted via [`ensure_models_locked`]; + /// two racing first calls may still both LOAD an engine from the verified + /// files (CPU-only) — `get_or_init` keeps one. + fn global() -> Result<&'static ocrs::OcrEngine, String> { + if let Some(e) = ENGINE.get() { + return Ok(e); + } + let dir = model_dir()?; + let det_path = dir.join("text-detection.rten"); + let rec_path = dir.join("text-recognition.rten"); + ensure_models_locked( + &[ + (DETECTION_MODEL_URL, &det_path, DETECTION_MODEL_SHA256), + (RECOGNITION_MODEL_URL, &rec_path, RECOGNITION_MODEL_SHA256), + ], + &download, + )?; + let load = |p: &Path| { + rten::Model::load_file(p).map_err(|e| format!("loading {}: {e}", p.display())) + }; + let detection = load_model_healing( + DETECTION_MODEL_URL, + &det_path, + DETECTION_MODEL_SHA256, + &download, + &load, + )?; + let recognition = load_model_healing( + RECOGNITION_MODEL_URL, + &rec_path, + RECOGNITION_MODEL_SHA256, + &download, + &load, + )?; + let engine = ocrs::OcrEngine::new(ocrs::OcrEngineParams { + detection_model: Some(detection), + recognition_model: Some(recognition), + ..Default::default() + }) + .map_err(|e| format!("initializing OCR engine: {e}"))?; + Ok(ENGINE.get_or_init(|| engine)) + } + + pub(super) struct LazyOcrs; + + impl OcrBackend for LazyOcrs { + fn recognize_rgb(&self, width: u32, height: u32, rgb: &[u8]) -> Result { + let engine = global()?; + let source = ocrs::ImageSource::from_bytes(rgb, (width, height)) + .map_err(|e| format!("preparing image: {e}"))?; + let input = engine + .prepare_input(source) + .map_err(|e| format!("preparing OCR input: {e}"))?; + engine + .get_text(&input) + .map_err(|e| format!("recognizing text: {e}")) + } + } + + /// Hermetic model-plumbing tests: fetch fns are injected, nothing touches + /// the network or rten. Named ocr_model_* so they ride the `ocr` filter. + #[cfg(test)] + mod tests { + use std::path::{Path, PathBuf}; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::sync::Arc; + + use super::*; + + const GOOD: &[u8] = b"pretend this is an rten model"; + /// sha256 of GOOD, computed in-test (no second pinning to drift). + fn good_sha() -> String { + use sha2::Digest as _; + format!("{:x}", sha2::Sha256::digest(GOOD)) + } + + /// Fresh scratch dir per test (std-only; no tempfile dev-dep). + struct Scratch(PathBuf); + impl Scratch { + fn new(tag: &str) -> Self { + let dir = std::env::temp_dir().join(format!( + "verity-ocr-model-tests-{}-{}-{tag}", + std::process::id(), + TMP_SEQ.fetch_add(1, Ordering::Relaxed) + )); + std::fs::create_dir_all(&dir).expect("scratch dir"); + Scratch(dir) + } + fn path(&self, name: &str) -> PathBuf { + self.0.join(name) + } + } + impl Drop for Scratch { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.0); + } + } + + fn counting_fetch( + counter: Arc, + body: &'static [u8], + ) -> impl Fn(&str, &Path) -> Result<(), String> + Sync { + move |_url: &str, dest: &Path| { + counter.fetch_add(1, Ordering::SeqCst); + std::fs::write(dest, body).map_err(|e| e.to_string()) + } + } + + #[test] + fn ocr_model_fetch_is_single_flight_two_threads_one_download() { + let scratch = Scratch::new("single-flight"); + let dest = scratch.path("model.rten"); + let sha = good_sha(); + let count = Arc::new(AtomicUsize::new(0)); + std::thread::scope(|s| { + for _ in 0..2 { + let fetch = counting_fetch(count.clone(), GOOD); + let (dest, sha) = (dest.clone(), sha.clone()); + s.spawn(move || { + ensure_models_locked(&[("mem://model", &dest, &sha)], &fetch) + .expect("fetch succeeds"); + }); + } + }); + assert_eq!( + count.load(Ordering::SeqCst), + 1, + "two racing first scans must produce exactly one download" + ); + assert_eq!(std::fs::read(&dest).unwrap(), GOOD); + } + + #[test] + fn ocr_model_corrupt_cache_self_heals_with_one_refetch() { + let scratch = Scratch::new("self-heal"); + let dest = scratch.path("model.rten"); + std::fs::write(&dest, b"bitrot").unwrap(); + let count = Arc::new(AtomicUsize::new(0)); + let fetch = counting_fetch(count.clone(), GOOD); + ensure_model_file("mem://model", &dest, &good_sha(), &fetch) + .expect("corrupt cache heals"); + assert_eq!(count.load(Ordering::SeqCst), 1, "exactly one refetch"); + assert_eq!( + std::fs::read(&dest).unwrap(), + GOOD, + "healed to canonical bytes" + ); + } + + #[test] + fn ocr_model_download_hash_mismatch_is_rejected_and_deleted() { + let scratch = Scratch::new("bad-download"); + let dest = scratch.path("model.rten"); + let count = Arc::new(AtomicUsize::new(0)); + let fetch = counting_fetch(count.clone(), b"not the pinned artifact"); + let err = ensure_model_file("mem://model", &dest, &good_sha(), &fetch) + .expect_err("mismatched download must be rejected"); + assert!(err.contains("checksum mismatch"), "typed reason: {err}"); + assert_eq!( + count.load(Ordering::SeqCst), + 1, + "refetched ONCE, then failed" + ); + assert!( + !dest.exists(), + "a mismatching file must never linger as cache" + ); + } + + #[test] + fn ocr_model_unloadable_file_is_deleted_refetched_once_then_loads() { + let scratch = Scratch::new("heal-load"); + let dest = scratch.path("model.rten"); + std::fs::write(&dest, GOOD).unwrap(); // checksum-valid but "unloadable" + let fetches = Arc::new(AtomicUsize::new(0)); + let fetch = counting_fetch(fetches.clone(), GOOD); + let loads = AtomicUsize::new(0); + let load = |p: &Path| -> Result, String> { + if loads.fetch_add(1, Ordering::SeqCst) == 0 { + Err("rten refused (test)".into()) + } else { + std::fs::read(p).map_err(|e| e.to_string()) + } + }; + let model = load_model_healing("mem://model", &dest, &good_sha(), &fetch, &load) + .expect("second load succeeds after heal"); + assert_eq!(model, GOOD); + assert_eq!(fetches.load(Ordering::SeqCst), 1, "healed with ONE refetch"); + assert_eq!(loads.load(Ordering::SeqCst), 2); + } + + #[test] + fn ocr_model_unloadable_after_refetch_fails_typed() { + let scratch = Scratch::new("heal-load-fails"); + let dest = scratch.path("model.rten"); + std::fs::write(&dest, GOOD).unwrap(); + let fetches = Arc::new(AtomicUsize::new(0)); + let fetch = counting_fetch(fetches.clone(), GOOD); + let load = |_: &Path| -> Result<(), String> { Err("rten refused (test)".into()) }; + let err = load_model_healing("mem://model", &dest, &good_sha(), &fetch, &load) + .expect_err("still-unloadable model fails typed"); + assert!(err.contains("after refetch"), "typed reason: {err}"); + assert_eq!( + fetches.load(Ordering::SeqCst), + 1, + "refetched ONCE, not in a loop" + ); + } + } +} + +#[cfg(not(feature = "ocr"))] +mod engine { + use super::OcrBackend; + + /// Stub for `--no-default-features` builds: every OCR attempt fails typed + /// and disclosed — never silently empty. + pub(super) struct LazyOcrs; + + impl OcrBackend for LazyOcrs { + fn recognize_rgb(&self, _w: u32, _h: u32, _rgb: &[u8]) -> Result { + Err("verity-server was built without the 'ocr' cargo feature".into()) + } + } +} diff --git a/crates/verity-server/src/testdata/ocr-sample.jpg b/crates/verity-server/src/testdata/ocr-sample.jpg new file mode 100644 index 0000000..52829c2 Binary files /dev/null and b/crates/verity-server/src/testdata/ocr-sample.jpg differ diff --git a/crates/verity-server/src/testdata/ocr-sample.png b/crates/verity-server/src/testdata/ocr-sample.png new file mode 100644 index 0000000..665dea9 Binary files /dev/null and b/crates/verity-server/src/testdata/ocr-sample.png differ diff --git a/ingest/tests/test_gdrive.py b/ingest/tests/test_gdrive.py index 3644cc2..eb5be30 100644 --- a/ingest/tests/test_gdrive.py +++ b/ingest/tests/test_gdrive.py @@ -1436,4 +1436,18 @@ def test_word_mimes_are_binary_extractable(): "application/vnd.openxmlformats-officedocument.wordprocessingml.document" ) assert is_binary_extractable("application/msword") - assert not is_binary_extractable("image/png") + + +def test_image_mimes_are_binary_extractable_png_jpeg_only(): + """The server's local OCR tier (extract.rs + ocr.rs) handles PNG and JPEG; + the mime gate must deliver exactly those. Formats the server would refuse + (GIF/TIFF/WebP/SVG) stay honestly metadata-only — pin both directions. + The sharepoint connector imports this gate, so this pins it there too.""" + from verity_ingest.connectors.gdrive import is_binary_extractable + + assert is_binary_extractable("image/png") + assert is_binary_extractable("image/jpeg") + assert not is_binary_extractable("image/gif") + assert not is_binary_extractable("image/tiff") + assert not is_binary_extractable("image/webp") + assert not is_binary_extractable("image/svg+xml") diff --git a/ingest/verity_ingest/connectors/gdrive.py b/ingest/verity_ingest/connectors/gdrive.py index 109aca9..3471d29 100644 --- a/ingest/verity_ingest/connectors/gdrive.py +++ b/ingest/verity_ingest/connectors/gdrive.py @@ -15,17 +15,20 @@ - Google Docs → ``files.export`` as ``text/plain`` - ``text/*`` and ``application/json`` → direct download (``alt=media``), delivered inline as ``content`` text -- PDF / PPTX / XLS(X) → direct download (``alt=media``), delivered as raw - bytes in ``content_base64`` (+ ``filename``); the SERVER runs the Tier-1 - extractor (verity-server extract.rs: Rust-native, deterministic, no OCR). +- PDF / Office formats / PNG / JPEG → direct download (``alt=media``), + delivered as raw bytes in ``content_base64`` (+ ``filename``); the SERVER + runs the Tier-1 extractor (verity-server extract.rs: Rust-native and local + — deterministic text layers, plus best-effort local OCR for scanned PDFs + and images, disclosed as ``pdf-ocr``/``image-ocr`` on the receipt). This was chosen over posting bytes to ``POST /v1/files`` because /v1/files writes under a scope handle whose principals would REPLACE the mirrored per-file ACL this connector computed — the whole point of a Tier-A connector. Riding the existing documents endpoint keeps one sink, the same visibility/entity mapping, and ACL-before-content ordering; the smallest - honest change. Typed extraction failures (encrypted PDF, scanned/image PDF - with no text layer, parse failure) land METADATA-ONLY server-side with the - reason disclosed on the stored record — never silently indexed as empty. + honest change. Typed extraction failures (encrypted PDF, scanned PDF where + OCR found nothing, OCR engine unavailable, parse failure) land + METADATA-ONLY server-side with the reason disclosed on the stored record — + never silently indexed as empty. - everything else → metadata + ACL only, no content bytes ACL mapping (fail-closed, §5e.6 / §6b): @@ -226,9 +229,9 @@ class GDriveDocumentEvent(DocumentEvent): # Binary formats the server's Tier-1 extractor handles (verity-server -# extract.rs): text-based PDF, PPTX, XLS(X). Deliberately NOT .doc/.docx or -# legacy .ppt — Google Docs already export as text, and anything else stays -# honestly metadata-only until a later tier. +# extract.rs): PDF (text layer, or the local OCR tier for scanned ones), +# Office formats, and PNG/JPEG images (local OCR). Deliberately NOT legacy +# .ppt, GIF/TIFF/WebP, etc. — anything else stays honestly metadata-only. BINARY_EXTRACTABLE_MIMES = frozenset( { "application/pdf", @@ -241,6 +244,11 @@ class GDriveDocumentEvent(DocumentEvent): "application/vnd.openxmlformats-officedocument.presentationml.presentation", "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", "application/vnd.ms-excel", + # Images: the server's local OCR tier (ocr.rs, ocrs+rten) extracts + # printed text best-effort, disclosed as method "image-ocr"; an image + # with no recognizable text lands metadata-only with a typed reason. + "image/png", + "image/jpeg", } )